diff --git a/.gitignore b/.gitignore index 0d9a62c8..0184dc8c 100644 --- a/.gitignore +++ b/.gitignore @@ -83,3 +83,4 @@ output/ /tests/test_data/wedowind/Vortex Generator/ # matplotlib testing **/result_images/ +/docs/v1/extra_docs/ diff --git a/benchmarking/__init__.py b/benchmarking/__init__.py new file mode 100644 index 00000000..5fa58eff --- /dev/null +++ b/benchmarking/__init__.py @@ -0,0 +1,4 @@ +"""Experimental v1 benchmarking work (synthetic datasets and candidate methods). + +Not part of the published ``res-wind-up`` package. +""" diff --git a/benchmarking/baselines/__init__.py b/benchmarking/baselines/__init__.py new file mode 100644 index 00000000..a2b99526 --- /dev/null +++ b/benchmarking/baselines/__init__.py @@ -0,0 +1,13 @@ +"""v0 binned power-curve baseline for the benchmarking harness. + +Wires wind_up's pre/post power-performance pipeline behind the harness's ``Method`` seam so +the existing v0 method can be scored against synthetic ground truth, establishing the +baseline every new method must beat. +""" + +from __future__ import annotations + +from benchmarking.baselines.hot_context import HotV0Context, build_hot_v0_context +from benchmarking.baselines.v0_binned import V0BinnedMethod + +__all__ = ["HotV0Context", "V0BinnedMethod", "build_hot_v0_context"] diff --git a/benchmarking/baselines/assets/HOT.yaml b/benchmarking/baselines/assets/HOT.yaml new file mode 100644 index 00000000..01db98d1 --- /dev/null +++ b/benchmarking/baselines/assets/HOT.yaml @@ -0,0 +1,28 @@ +# Vendored from resgroup/hill-of-towie-open-source-analysis +# scripts/uplift_analysis_2025/wind_up_config/asset/HOT.yaml +# Kept in-repo so the v0 baseline runs without depending on that external repo. +name: Hill of Towie +wtgs: + - T01 + - T02 + - T03 + - T04 + - T05 + - T06 + - T07 + - T08 + - T09 + - T10 + - T11 + - T12 + - T13 + - T14 + - T15 + - T16 + - T17 + - T18 + - T19 + - T20 + - T21 +turbine_types: + - !include turbine_type/SWT_2p3_82.yaml diff --git a/benchmarking/baselines/assets/optimized_northing_corrections.yaml b/benchmarking/baselines/assets/optimized_northing_corrections.yaml new file mode 100644 index 00000000..0562a4fe --- /dev/null +++ b/benchmarking/baselines/assets/optimized_northing_corrections.yaml @@ -0,0 +1,53 @@ +# Vendored from resgroup/hill-of-towie-open-source-analysis +# scripts/uplift_analysis_2025/wind_up_config/northing/optimized_northing_corrections.yaml +# These corrections go back to 2016, matching the stable no-upgrade window the v0 baseline +# uses; the more recent wfc_analysis_2026 northing file does not reach that far back. + - ['T01', 2016-01-01 00:00:00, 156.77226715087892] + - ['T01', 2016-01-22 14:00:00, 154.53022384643555] + - ['T01', 2016-02-22 06:10:00, 156.88454132080076] + - ['T01', 2016-03-20 15:50:00, 155.695051574707] + - ['T01', 2016-04-21 16:50:00, 156.87261695861815] + - ['T01', 2016-06-02 07:20:00, -25.836227035522455] + - ['T01', 2017-04-23 22:40:00, -4.762451171875] + - ['T01', 2017-05-04 11:00:00, -23.675787353515602] + - ['T02', 2016-01-01 00:00:00, 171.06498107910159] + - ['T02', 2016-06-02 11:10:00, 8.570095825195295] + - ['T03', 2016-01-01 00:00:00, -1.8987007141113281] + - ['T04', 2016-01-01 00:00:00, -7.872805786132801] + - ['T05', 2016-01-01 00:00:00, -3.5416046142578352] + - ['T05', 2017-05-03 06:30:00, 32.26841430664061] + - ['T05', 2018-04-21 14:20:00, 13.180883026123098] + - ['T06', 2016-01-01 00:00:00, -7.327435302734386] + - ['T07', 2016-01-01 00:00:00, -16.487994384765614] + - ['T08', 2016-01-01 00:00:00, -4.101350402832026] + - ['T08', 2020-07-27 11:10:00, -2.4224990844726335] + - ['T09', 2016-01-01 00:00:00, 13.869790649414085] + - ['T10', 2016-01-01 00:00:00, -22.53185272216797] + - ['T10', 2016-02-11 10:10:00, -30.683580017089866] + - ['T10', 2016-03-11 05:50:00, -21.081747436523415] + - ['T11', 2016-01-01 00:00:00, -20.928597640991228] + - ['T11', 2019-08-19 08:40:00, -26.41037063598634] + - ['T11', 2021-03-18 13:00:00, -31.436821079254173] + - ['T11', 2021-07-19 02:20:00, -22.862120819091842] + - ['T11', 2023-12-06 17:30:00, -20.633160400390636] + - ['T12', 2016-01-01 00:00:00, -19.703698730468773] + - ['T12', 2016-04-20 13:20:00, -142.98569173812868] + - ['T12', 2016-06-02 10:00:00, 0.538665771484375] + - ['T12', 2016-09-18 08:40:00, -162.54795513153076] + - ['T12', 2020-06-18 12:30:00, 7.401036834716791] + - ['T13', 2016-01-01 00:00:00, 1.261445617675804] + - ['T14', 2016-01-01 00:00:00, -11.315747833251976] + - ['T15', 2016-01-01 00:00:00, -15.957545471191395] + - ['T16', 2016-01-01 00:00:00, -17.773095703124966] + - ['T16', 2017-05-19 01:00:00, 80.36924819946289] + - ['T16', 2017-06-18 08:00:00, 89.37665023803709] + - ['T16', 2017-08-09 10:20:00, 82.41613636016845] + - ['T16', 2023-02-10 10:00:00, -6.421942138671852] + - ['T17', 2016-01-01 00:00:00, -15.991704559326166] + - ['T17', 2019-08-10 23:00:00, -14.783495330810524] + - ['T18', 2016-01-01 00:00:00, -2.3910442352294297] + - ['T19', 2016-01-01 00:00:00, 8.268119049072254] + - ['T19', 2019-07-12 10:00:00, 107.24801597595213] + - ['T19', 2019-12-24 09:00:00, -15.878029632568314] + - ['T20', 2016-01-01 00:00:00, -1.6393432617186932] + - ['T21', 2016-01-01 00:00:00, 4.474188232421824] diff --git a/benchmarking/baselines/assets/turbine_type/SWT_2p3_82.yaml b/benchmarking/baselines/assets/turbine_type/SWT_2p3_82.yaml new file mode 100644 index 00000000..b5d715d9 --- /dev/null +++ b/benchmarking/baselines/assets/turbine_type/SWT_2p3_82.yaml @@ -0,0 +1,7 @@ +# Vendored from resgroup/hill-of-towie-open-source-analysis +# scripts/uplift_analysis_2025/wind_up_config/asset/turbine_type/SWT_2p3_82.yaml +turbine_type: SWT-2.3-82 +rated_power_kw: 2300 +rotor_diameter_m: 82 +normal_operation_pitch_range: [-10,40] +normal_operation_genrpm_range: [500,1600] diff --git a/benchmarking/baselines/block_bootstrap.py b/benchmarking/baselines/block_bootstrap.py new file mode 100644 index 00000000..3906ef10 --- /dev/null +++ b/benchmarking/baselines/block_bootstrap.py @@ -0,0 +1,338 @@ +"""Circular block bootstrap for a toggle energy-ratio uplift. + +The uncertainty behind ``ToggleSpecialistMethod``, kept separate because it says nothing about +SCADA: it takes paired ``(test, reference)`` sums on a timeline and returns a sigma. + +A block is a wall-clock interval carrying its on- and off-rows **together**, so the on/off pairing +that makes the estimate precise survives resampling. Blocks start anywhere on the timebase grid and +wrap past the campaign end. Both ``rho_up`` and ``rho_base`` are recomputed per resample and the +ratio re-formed, rather than linearised. + +Block sums come from prefix sums over the records (doubled end-to-end so a wrapped block is still +two lookups), so a resample is a gather-and-subtract rather than a pass over the data. + +Rationale for the design and for the block length: findings F28. +""" + +from __future__ import annotations + +import math +from dataclasses import dataclass +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd +from scipy.stats import norm, t + +if TYPE_CHECKING: + from collections.abc import Mapping + + import numpy.typing as npt + +# Resamples per chunk; keeps the (chunk, n_draw, n_cells, 4) gather to tens of MB. +_RESAMPLE_CHUNK = 250 +# Per (cell, segment): the numerator and denominator of each segment's rho. +_N_QUANTITIES = 4 +_MIN_RESAMPLES_FOR_SPREAD = 2 +# With one block covering the whole campaign, every resample is that campaign: nothing varies. +_MIN_BLOCKS_FOR_SPREAD = 2 +# Normal -/+1 sigma percentiles, so (p84 - p16) / 2 is a sigma for a normal. +_SIGMA_PERCENTILES = (100.0 * norm.cdf(-1.0), 100.0 * norm.cdf(1.0)) +# Floor on the fallback's degrees of freedom, mirroring wind_up's `clip(lower=2)`: with one +# record a side has no scatter of its own, so df=1 is the widest t the convention allows. +_MIN_FALLBACK_DF = 1 +# Fewest records a segment needs before its own scatter is worth measuring, for `relative_scatter`. +_MIN_RECORDS_PER_SIDE = 3 +# Records per side over which the report ramps linearly from the fallback to the bootstrap (F33): +# pure fallback at/below LO, pure bootstrap at/above HI. +_BLEND_LO_RECORDS = 3 +_BLEND_HI_RECORDS = 7 + + +@dataclass(frozen=True) +class CellUncertainty: + """The uncertainty of one cell's uplift (the headline, or one condition bin). + + :param sigma: the reported 1-sigma: a records-weighted ramp from ``sigma_fallback`` (sparse cell) + to ``sigma_bootstrap`` (well-populated cell) + :param sigma_bootstrap: std of the resampled uplifts (``ddof=1``); NaN when there were fewer than + two finite resamples + :param sigma_fallback: the t-inflated per-record-scatter estimate; finite even for a sparse cell + :param sigma_robust: ``(p84 - p16) / 2`` of the resampled uplifts + :param frac_resamples_finite: fraction of resamples with a finite uplift; below 1 means some + resamples drew no baseline rows for this cell + """ + + sigma: float + sigma_bootstrap: float + sigma_fallback: float + sigma_robust: float + frac_resamples_finite: float + + +@dataclass(frozen=True) +class BootstrapResult: + """Per-cell uncertainties from one circular block bootstrap. + + :param n_blocks: blocks drawn per resample (``ceil(campaign / block)``) + :param cells: uncertainty per cell name, keyed as the caller keyed ``cell_membership`` + """ + + n_blocks: int + cells: dict[str, CellUncertainty] + + +def _nan_cells(names: list[str]) -> dict[str, CellUncertainty]: + """Return an all-NaN uncertainty for every cell (a campaign the bootstrap cannot run on).""" + nan = float("nan") + return { + name: CellUncertainty( + sigma=nan, sigma_bootstrap=nan, sigma_fallback=nan, sigma_robust=nan, frac_resamples_finite=nan + ) + for name in names + } + + +def relative_scatter( + test_power: npt.NDArray[np.float64], + ref_total: npt.NDArray[np.float64], + *, + upgraded: npt.NDArray[np.bool_], + baseline: npt.NDArray[np.bool_], +) -> float: + """Return the campaign's per-record relative scatter about its own test/reference ratio. + + ``sqrt(sum(y - R*x)^2 / sum(R*x)^2)`` per segment, pooled. A ratio of sums rather than a mean of + per-record ratios, because dividing by each record's predicted power explodes near cut-in, which + is exactly where the sparsest bins live. + + Measured over the whole campaign (thousands of records) so it is precise, then applied to a cell + with that cell's own record count — which is what lets a 1-record cell get a sigma at all. + """ + residual_sq = 0.0 + predicted_sq = 0.0 + for segment in (upgraded, baseline): + y, x = test_power[segment], ref_total[segment] + denom = x.sum() + if len(y) < _MIN_RECORDS_PER_SIDE or denom == 0: + continue + predicted = (y.sum() / denom) * x + residual_sq += float(((y - predicted) ** 2).sum()) + predicted_sq += float((predicted**2).sum()) + if predicted_sq <= 0: + return float("nan") + return math.sqrt(residual_sq / predicted_sq) + + +def _fallback_sigma(*, n_on: int, n_off: int, s_rel: float) -> float: + """Return a t-inflated per-record-scatter uncertainty for one cell (F33). + + ``s_rel * sqrt(1/n_on + 1/n_off)`` is the standard ratio-estimator error under multiplicative + per-record noise, propagated through ``rho_up / rho_base``. The ``scipy.stats.t`` multiplier is + ``wind_up``'s own convention (``pp_analysis``): ``t.ppf(norm.cdf(1), df)`` is the 1-sigma-equivalent + quantile, tending to 1.0 as ``df`` grows and widening as data runs out. ``df`` keys off the + *thinner* side, as ``wind_up`` does, because either side starves the ratio. + """ + if not math.isfinite(s_rel) or n_on < 1 or n_off < 1: + return float("nan") + df = max(min(n_on, n_off) - 1, _MIN_FALLBACK_DF) + return s_rel * math.sqrt(1.0 / n_on + 1.0 / n_off) * float(t.ppf(norm.cdf(1.0), df)) + + +def bootstrap_ratio_uplift( + *, + times: pd.DatetimeIndex, + test_power: npt.NDArray[np.float64], + ref_total: npt.NDArray[np.float64], + upgraded: npt.NDArray[np.bool_], + baseline: npt.NDArray[np.bool_], + cell_membership: Mapping[str, npt.NDArray[np.bool_]], + campaign_start: pd.Timestamp, + campaign_end: pd.Timestamp, + timebase: pd.Timedelta, + block_hours: float, + n_resamples: int, + seed: int, +) -> BootstrapResult: + """Bootstrap the 1-sigma uncertainty of ``rho_up / rho_base - 1`` for every cell. + + All array arguments are parallel over the **used** records only (the rows the point estimate + summed), in any order; they are sorted here. + + :param times: the used records' timestamps + :param test_power: the test turbine's power per used record + :param ref_total: the summed reference power per used record + :param upgraded: which used records are toggle-on + :param baseline: which used records are toggle-off + :param cell_membership: cell name -> which used records belong to it (the headline, or one + condition bin). Must be fixed by the point estimate so resampling cannot move a record + between bins. + :param campaign_start: the campaign's first timestamp; blocks tile forward from here, so gaps in + the used records are covered rather than closed up + :param campaign_end: the campaign's last timestamp + :param timebase: analysis timebase; sets the candidate block-start grid + :param block_hours: block length (F28). A length at or beyond the campaign leaves one block, which + nothing can vary, so every cell reports NaN rather than a spurious near-zero sigma. + :param n_resamples: resamples to draw + :param seed: RNG seed, so a reported sigma is reproducible + """ + names = list(cell_membership) + n_records = len(times) + campaign_s = (campaign_end - campaign_start) / pd.Timedelta(seconds=1) + timebase.total_seconds() + if n_records == 0 or campaign_s <= 0 or n_resamples < _MIN_RESAMPLES_FOR_SPREAD: + return BootstrapResult(n_blocks=0, cells=_nan_cells(names)) + + # Elapsed seconds via pandas rather than numpy datetime arithmetic: a tz-aware DatetimeIndex + # converts to an object array, which numpy cannot subtract. + elapsed = np.asarray((times - campaign_start) / pd.Timedelta(seconds=1), dtype=float) + order = np.argsort(elapsed, kind="stable") + seconds = elapsed[order] + prefix = _prefix_sums( + test_power=test_power[order], + ref_total=ref_total[order], + upgraded=upgraded[order], + baseline=baseline[order], + cell_membership={name: mask[order] for name, mask in cell_membership.items()}, + ) + + # Doubled timeline: a wrapped block [s, s+L) is then a plain contiguous range over `doubled`, + # so it needs no special case and stays two lookups. + doubled = np.concatenate([seconds, seconds + campaign_s]) + block_s = min(float(block_hours) * 3600.0, campaign_s) + starts = np.arange(0.0, campaign_s, timebase.total_seconds()) + lo_idx = np.searchsorted(doubled, starts, side="left") + hi_idx = np.searchsorted(doubled, starts + block_s, side="left") + n_blocks = math.ceil(campaign_s / block_s) + + rng = np.random.default_rng(seed) + totals = np.zeros((n_resamples, len(names), _N_QUANTITIES)) + for lo in range(0, n_resamples, _RESAMPLE_CHUNK): + hi = min(lo + _RESAMPLE_CHUNK, n_resamples) + drawn = rng.integers(0, len(starts), size=(hi - lo, n_blocks)) + totals[lo:hi] = (prefix[hi_idx[drawn]] - prefix[lo_idx[drawn]]).sum(axis=1) + + # The fallback is computed for every cell regardless, so a caller can re-judge the blend rule + # offline from a saved sweep instead of re-running one. + s_rel = relative_scatter(test_power, ref_total, upgraded=upgraded, baseline=baseline) + counts = { + name: (int((mask & upgraded).sum()), int((mask & baseline).sum())) for name, mask in cell_membership.items() + } + fallback = {name: _fallback_sigma(n_on=on, n_off=off, s_rel=s_rel) for name, (on, off) in counts.items()} + + if n_blocks < _MIN_BLOCKS_FOR_SPREAD: + # One block spans the whole campaign, so every resample is the same campaign and nothing can + # vary. Any sigma it returned would be float residue (~1e-15), not a real certainty. + return BootstrapResult(n_blocks=n_blocks, cells=_fallback_only_cells(names, fallback=fallback)) + + uplift = _uplift_from_totals(totals) + boot_weight = {name: _bootstrap_weight(min(on, off)) for name, (on, off) in counts.items()} + return BootstrapResult( + n_blocks=n_blocks, + cells=_summarise(uplift, names=names, fallback=fallback, boot_weight=boot_weight), + ) + + +def _bootstrap_weight(n_min: int) -> float: + """Weight on the bootstrap for a cell with ``n_min`` records on its thinner side (F33).""" + span = _BLEND_HI_RECORDS - _BLEND_LO_RECORDS + return float(np.clip((n_min - _BLEND_LO_RECORDS) / span, 0.0, 1.0)) + + +def _fallback_only_cells(names: list[str], *, fallback: dict[str, float]) -> dict[str, CellUncertainty]: + """Cells for a campaign the bootstrap cannot run on at all: the fallback is all there is.""" + nan = float("nan") + return { + name: CellUncertainty( + sigma=fallback[name], + sigma_bootstrap=nan, + sigma_fallback=fallback[name], + sigma_robust=nan, + frac_resamples_finite=nan, + ) + for name in names + } + + +def _prefix_sums( + *, + test_power: npt.NDArray[np.float64], + ref_total: npt.NDArray[np.float64], + upgraded: npt.NDArray[np.bool_], + baseline: npt.NDArray[np.bool_], + cell_membership: Mapping[str, npt.NDArray[np.bool_]], +) -> npt.NDArray[np.float64]: + """Cumulative ``(test, ref)`` sums per ``(cell, segment)`` over the doubled record timeline. + + Returns shape ``(2 * n_records + 1, n_cells, 4)``, where the last axis is + ``(test_on, ref_on, test_off, ref_off)`` and the leading zero row lets any block's sums be one + subtraction. Doubling the records mirrors the doubled timeline so a wrapped block reads + contiguously. + """ + names = list(cell_membership) + values = np.zeros((len(test_power), len(names), _N_QUANTITIES)) + for i, name in enumerate(names): + member = cell_membership[name] + on = member & upgraded + off = member & baseline + values[:, i, 0] = np.where(on, test_power, 0.0) + values[:, i, 1] = np.where(on, ref_total, 0.0) + values[:, i, 2] = np.where(off, test_power, 0.0) + values[:, i, 3] = np.where(off, ref_total, 0.0) + doubled = np.concatenate([values, values], axis=0) + return np.concatenate([np.zeros((1, len(names), _N_QUANTITIES)), np.cumsum(doubled, axis=0)], axis=0) + + +def _uplift_from_totals(totals: npt.NDArray[np.float64]) -> npt.NDArray[np.float64]: + """Re-form ``rho_up / rho_base - 1`` per (resample, cell) from resampled sums. + + The degeneracy guards mirror the point estimate's, so a resample fails only where the point + estimate would have failed on the same rows. + """ + test_on, ref_on, test_off, ref_off = (totals[..., k] for k in range(_N_QUANTITIES)) + nan = np.full(test_on.shape, np.nan) + rho_up = np.divide(test_on, ref_on, out=nan.copy(), where=ref_on != 0) + rho_base = np.divide(test_off, ref_off, out=nan.copy(), where=ref_off != 0) + valid = np.isfinite(rho_base) & (rho_base != 0) & np.isfinite(rho_up) + return np.divide(rho_up, rho_base, out=nan.copy(), where=valid) - 1.0 + + +def _summarise( + uplift: npt.NDArray[np.float64], + *, + names: list[str], + fallback: dict[str, float], + boot_weight: dict[str, float], +) -> dict[str, CellUncertainty]: + """Reduce each cell's resamples to a bootstrap sigma, and ramp between it and the fallback. + + The reported sigma is ``w*bootstrap + (1-w)*fallback`` where ``w`` is ``boot_weight`` (0 at the + sparse end, 1 once the cell is well populated), or whichever component is finite when the other is + not. The bootstrap reports NaN when there are fewer than two finite resamples. + """ + cells = {} + for i, name in enumerate(names): + values = uplift[:, i] + finite = np.isfinite(values) + n_finite = int(finite.sum()) + frac = n_finite / len(values) + kept = values[finite] + usable = n_finite >= _MIN_RESAMPLES_FOR_SPREAD + boot = float(kept.std(ddof=1)) if usable else float("nan") + robust = float(np.subtract(*np.percentile(kept, _SIGMA_PERCENTILES[::-1])) / 2.0) if usable else float("nan") + cells[name] = CellUncertainty( + sigma=_blend(bootstrap=boot, fallback=fallback[name], weight=boot_weight[name]), + sigma_bootstrap=boot, + sigma_fallback=fallback[name], + sigma_robust=robust, + frac_resamples_finite=frac, + ) + return cells + + +def _blend(*, bootstrap: float, fallback: float, weight: float) -> float: + """Linear ramp ``weight*bootstrap + (1-weight)*fallback``, using whichever side is finite.""" + if math.isfinite(bootstrap) and math.isfinite(fallback): + return weight * bootstrap + (1.0 - weight) * fallback + if math.isfinite(bootstrap): + return bootstrap + return fallback diff --git a/benchmarking/baselines/era5_derived.py b/benchmarking/baselines/era5_derived.py new file mode 100644 index 00000000..0e73202d --- /dev/null +++ b/benchmarking/baselines/era5_derived.py @@ -0,0 +1,179 @@ +"""Derive physically meaningful quantities from raw (synced) ERA5 columns (shared, Issue 9). + +The methods consume ERA5 as raw Open-Meteo columns; this module turns those into the derived, +treatment-invariant quantities that actually drive turbine power and its scatter, so every method +*and* the CEM matching step share one implementation: + +* ``shear_exponent`` — the power-law exponent ``alpha = ln(ws_100m/ws_10m)/ln(100/10)``; folds the + collinear 10 m / 100 m speeds into one physical vertical-shear (stability) signal. +* ``wind_speed_hub`` — hub-height wind speed by the shear power law, + ``ws_hh = ws_100m * (hub_height_m/100)^alpha`` (needs the site's hub height). +* ``gust_ratio`` — ``wind_gusts_10m / wind_speed_10m``, a unitless TI-like turbulence proxy + (NaN below a calm-wind floor where the ratio degenerates). +* ``veer`` — vertical direction veer ``wind_direction_100m - wind_direction_10m`` wrapped to + ±180°. +* ``air_density`` — moist-air density from 2 m temperature, surface pressure and relative + humidity (partial pressures of dry air and vapour, Magnus saturation formula). + +All functions are NaN-tolerant (LightGBM handles NaN natively) and operate on the *aligned* ERA5 +frame the lag sync produces (original Open-Meteo column names). +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd + +if TYPE_CHECKING: + from collections.abc import Sequence + +# The derivation vocabulary: each name is both the config token and the output column name. +ERA5_DERIVATIONS: tuple[str, ...] = ( + "shear_exponent", + "wind_speed_hub", + "gust_ratio", + "gust_margin", + "veer", + "air_density", +) + +# The two ERA5 wind-speed levels the shear exponent is fit between. +_SHEAR_LO_M = 10.0 +_SHEAR_HI_M = 100.0 +# Below this 10 m wind speed the gust ratio degenerates (tiny denominator) and TI is meaningless. +_GUST_RATIO_MIN_WS = 1.0 +# Specific gas constants [J/(kg K)] for dry air and water vapour. +_R_DRY = 287.05 +_R_VAPOUR = 461.5 +_KELVIN_OFFSET = 273.15 +# Magnus saturation-vapour-pressure constants (over water), e_s in Pa for t in degC. +_MAGNUS_A = 611.2 +_MAGNUS_B = 17.62 +_MAGNUS_C = 243.12 + + +def shear_exponent(ws_lo: pd.Series, ws_hi: pd.Series) -> pd.Series: + """Power-law shear exponent ``alpha = ln(ws_hi/ws_lo)/ln(hi/lo)``; NaN where either speed <= 0.""" + lo = ws_lo.to_numpy(dtype=float) + hi = ws_hi.to_numpy(dtype=float) + positive = (lo > 0) & (hi > 0) + alpha = np.full(len(lo), np.nan) + alpha[positive] = np.log(hi[positive] / lo[positive]) / np.log(_SHEAR_HI_M / _SHEAR_LO_M) + return pd.Series(alpha, index=ws_hi.index, name="shear_exponent") + + +def hub_height_wind_speed(ws_hi: pd.Series, alpha: pd.Series, *, hub_height_m: float) -> pd.Series: + """Interpolate to hub height with the shear power law: ``ws_hh = ws_hi * (hh/hi)^alpha``.""" + hi = ws_hi.to_numpy(dtype=float) + a = alpha.to_numpy(dtype=float) + with np.errstate(invalid="ignore"): + ws_hh = hi * (hub_height_m / _SHEAR_HI_M) ** a + return pd.Series(ws_hh, index=ws_hi.index, name="wind_speed_hub") + + +def gust_ratio(gusts: pd.Series, ws: pd.Series, *, min_ws: float = _GUST_RATIO_MIN_WS) -> pd.Series: + """TI-like gust ratio ``gusts/ws``; NaN where ``ws < min_ws`` (calm-wind degenerate denominator).""" + g = gusts.to_numpy(dtype=float) + w = ws.to_numpy(dtype=float) + valid = w >= min_ws + ratio = np.divide(g, w, out=np.full(len(w), np.nan), where=valid) + return pd.Series(ratio, index=ws.index, name="gust_ratio") + + +def gust_margin(gusts: pd.Series, ws: pd.Series) -> pd.Series: + """Gust margin ``gusts - ws`` [m/s] — an absolute gustiness signal. + + On Hill of Towie this correlates with measured nacelle TI better than the ratio form + (the calm-wind denominator makes ``gust_ratio`` nearly uncorrelated with TI at 10 min). + """ + margin = gusts.to_numpy(dtype=float) - ws.to_numpy(dtype=float) + return pd.Series(margin, index=ws.index, name="gust_margin") + + +def vertical_veer(wd_hi: pd.Series, wd_lo: pd.Series) -> pd.Series: + """Vertical direction veer ``wd_hi - wd_lo`` wrapped to (-180, 180] degrees.""" + diff = wd_hi.to_numpy(dtype=float) - wd_lo.to_numpy(dtype=float) + wrapped = -((180.0 - diff) % 360.0 - 180.0) + return pd.Series(wrapped, index=wd_hi.index, name="veer") + + +def air_density(temperature_c: pd.Series, pressure_hpa: pd.Series, relative_humidity_pct: pd.Series) -> pd.Series: + """Moist-air density [kg/m3] from 2 m temperature [degC], surface pressure [hPa] and RH [%]. + + Partial-pressure form: ``rho = p_dry/(R_dry T) + p_vapour/(R_vapour T)`` with the vapour + pressure from the Magnus saturation formula scaled by relative humidity. + """ + t_c = temperature_c.to_numpy(dtype=float) + t_k = t_c + _KELVIN_OFFSET + p_pa = pressure_hpa.to_numpy(dtype=float) * 100.0 + saturation_pa = _MAGNUS_A * np.exp(_MAGNUS_B * t_c / (_MAGNUS_C + t_c)) + p_vapour = np.clip(relative_humidity_pct.to_numpy(dtype=float), 0.0, 100.0) / 100.0 * saturation_pa + p_dry = p_pa - p_vapour + rho = p_dry / (_R_DRY * t_k) + p_vapour / (_R_VAPOUR * t_k) + return pd.Series(rho, index=temperature_c.index, name="air_density") + + +# Raw Open-Meteo columns each derivation needs (validated before deriving so a missing input is a +# clear configuration error, not a KeyError deep in numpy). +_REQUIRED_RAW: dict[str, tuple[str, ...]] = { + "shear_exponent": ("wind_speed_10m", "wind_speed_100m"), + "wind_speed_hub": ("wind_speed_10m", "wind_speed_100m"), + "gust_ratio": ("wind_gusts_10m", "wind_speed_10m"), + "gust_margin": ("wind_gusts_10m", "wind_speed_100m"), + "veer": ("wind_direction_100m", "wind_direction_10m"), + "air_density": ("temperature_2m", "surface_pressure", "relative_humidity_2m"), +} + + +def era5_derived_frame( + aligned_era5: pd.DataFrame, + *, + derivations: Sequence[str], + hub_height_m: float | None = None, +) -> pd.DataFrame: + """Build the requested derived columns from an aligned ERA5 frame (one column per derivation). + + :param aligned_era5: ERA5 aligned to the analysis grid (original Open-Meteo column names) + :param derivations: which of :data:`ERA5_DERIVATIONS` to compute (also the output column names) + :param hub_height_m: turbine hub height; required by the ``wind_speed_hub`` derivation + """ + unknown = [d for d in derivations if d not in ERA5_DERIVATIONS] + if unknown: + msg = f"unknown ERA5 derivation(s) {unknown}; available: {list(ERA5_DERIVATIONS)}" + raise ValueError(msg) + if "wind_speed_hub" in derivations and hub_height_m is None: + msg = "the wind_speed_hub derivation requires hub_height_m (the site's turbine hub height)." + raise ValueError(msg) + missing = sorted({c for d in derivations for c in _REQUIRED_RAW[d]} - set(aligned_era5.columns)) + if missing: + msg = f"aligned ERA5 is missing columns {missing} required by derivations {list(derivations)}" + raise ValueError(msg) + + # alpha feeds both shear_exponent and wind_speed_hub; compute it at most once. + alpha: pd.Series | None = None + if {"shear_exponent", "wind_speed_hub"} & set(derivations): + alpha = shear_exponent(aligned_era5["wind_speed_10m"], aligned_era5["wind_speed_100m"]) + out = pd.DataFrame(index=aligned_era5.index) + for name in derivations: + if name == "shear_exponent": + assert alpha is not None # noqa: S101 - set above whenever this branch is reachable + out[name] = alpha + elif name == "wind_speed_hub": + assert alpha is not None # noqa: S101 - set above whenever this branch is reachable + assert hub_height_m is not None # noqa: S101 - narrowed above; mypy needs the hint + out[name] = hub_height_wind_speed(aligned_era5["wind_speed_100m"], alpha, hub_height_m=hub_height_m) + elif name == "gust_ratio": + out[name] = gust_ratio(aligned_era5["wind_gusts_10m"], aligned_era5["wind_speed_10m"]) + elif name == "gust_margin": + out[name] = gust_margin(aligned_era5["wind_gusts_10m"], aligned_era5["wind_speed_100m"]) + elif name == "veer": + out[name] = vertical_veer(aligned_era5["wind_direction_100m"], aligned_era5["wind_direction_10m"]) + elif name == "air_density": + out[name] = air_density( + aligned_era5["temperature_2m"], + aligned_era5["surface_pressure"], + aligned_era5["relative_humidity_2m"], + ) + return out diff --git a/benchmarking/baselines/era5_sync.py b/benchmarking/baselines/era5_sync.py new file mode 100644 index 00000000..39754b5b --- /dev/null +++ b/benchmarking/baselines/era5_sync.py @@ -0,0 +1,167 @@ +"""Align hourly ERA5 reanalysis to the 10-min SCADA grid (shared across benchmarking methods). + +ERA5 arrives hourly; SCADA is 10-min. Two steps, adapted from the logic in +``wind_up.reanalysis_data`` (``_reanalysis_upsample`` / ``_find_best_shift_and_corr``, +which are private there) but kept local so the methods stay v0-independent: + +1. :func:`upsample_era5_to_timebase` resamples ERA5 onto the analysis timebase and + forward-fills within each hour. **Every raw column is passed through under its original + Open-Meteo name** (no renaming); for back-compat with the R-learner and the shared + diagnostics, neutral ``era5_ws`` / ``era5_wd`` *aliases* of wind speed / direction are + added alongside the raw columns. +2. :func:`find_best_lag` sweeps the integer row-shift that maximises the correlation + between ERA5 wind speed and a reference (wind-farm) wind speed, recovering the lag + between the reanalysis and the site. + +:func:`sync_era5` combines both and returns the aligned ERA5 (all columns), plus the chosen +lag and the correlation-vs-lag sweep for diagnostics. + +This module was promoted out of ``benchmarking.baselines.rlearner.era5_sync`` once a second +method (``power_model``) needed it; that module now re-exports from here for back-compat. +""" + +from __future__ import annotations + +import math +from dataclasses import dataclass + +import numpy as np +import pandas as pd + +_MIN_OVERLAP = 3 + +# Neutral, source-agnostic aliases (no wind_up / v0 vocabulary) added alongside the raw columns +# so the R-learner's ``era5_features`` and the shared diagnostics keep a stable ws/wd handle. +ERA5_WS = "era5_ws" +ERA5_WD = "era5_wd" + +# Open-Meteo raw column names ERA5 ships with (see ``wind_up.era5``). +_RAW_WS = "wind_speed_100m" +_RAW_WD = "wind_direction_100m" + +_DEFAULT_MAX_LAG = pd.Timedelta(hours=24) + + +@dataclass +class Era5SyncResult: + """ERA5 aligned to a target index, with the recovered lag and the sweep. + + :param aligned: frame indexed by the target index with **all** raw ERA5 columns (original + Open-Meteo names) plus the :data:`ERA5_WS` / :data:`ERA5_WD` aliases (lag applied) + :param best_lag_rows: the integer row shift applied to ERA5 (positive = ERA5 shifted + forward to align with a lagging site signal) + :param best_corr: the wind-speed correlation at ``best_lag_rows`` + :param sweep: the correlation-vs-lag table (columns ``shift_rows``, ``corr``) + """ + + aligned: pd.DataFrame + best_lag_rows: int + best_corr: float + sweep: pd.DataFrame + + +def upsample_era5_to_timebase(era5_hourly_df: pd.DataFrame, *, timebase: pd.Timedelta) -> pd.DataFrame: + """Resample hourly ERA5 onto ``timebase`` and forward-fill within each source step. + + Mirrors ``wind_up.reanalysis_data._reanalysis_upsample``: resample with ``last``, extend + the index so the final source step's trailing slots exist, then forward-fill up to one + source step. **All** raw columns are kept under their original names; ``era5_ws`` / + ``era5_wd`` aliases of wind speed / direction are added for back-compat. + """ + source_step = pd.Timedelta(pd.Series(era5_hourly_df.index).diff().median()) + upsample_factor = round(source_step / timebase) + resampled = era5_hourly_df.resample(timebase, label="left").last() + if upsample_factor > 1: + tail = pd.DataFrame( + index=pd.date_range( + start=resampled.index[-1] + timebase, + periods=upsample_factor - 1, + freq=timebase, + ) + ) + resampled = pd.concat([resampled, tail]) + resampled = resampled.ffill(limit=upsample_factor - 1) + missing = {_RAW_WS, _RAW_WD} - set(era5_hourly_df.columns) + if missing: + msg = f"ERA5 frame is missing expected columns {sorted(missing)}; have {list(era5_hourly_df.columns)}" + raise ValueError(msg) + out = resampled.copy() + out[ERA5_WS] = resampled[_RAW_WS] + out[ERA5_WD] = resampled[_RAW_WD] + out.index.name = era5_hourly_df.index.name + return out + + +def find_best_lag( + *, + reference_ws: pd.Series, + era5_ws: pd.Series, + timebase: pd.Timedelta, + max_lag: pd.Timedelta = _DEFAULT_MAX_LAG, +) -> tuple[int, float, pd.DataFrame]: + """Find the integer row shift of ERA5 wind speed that best correlates with ``reference_ws``. + + Both series must share the analysis-grid index. Sweeps shifts in ``±max_lag`` (in steps of + ~10 min of rows, like wind_up) and returns the shift maximising ``corr(era5_ws.shift(s), + reference_ws)`` — so a positive shift advances ERA5 to meet a site signal that lags it. + """ + rows_per_hour = pd.Timedelta(hours=1) / timebase + # cap the sweep so the most extreme shift still overlaps the data (avoids empty slices) + max_rows = min(round(max_lag / timebase), len(era5_ws) - _MIN_OVERLAP) + step = max(1, math.ceil(rows_per_hour / 6)) + shifts = list(range(-max_rows, max_rows + 1, step)) + # a shift is only meaningful if a substantial chunk of data still overlaps; otherwise a + # handful of coincidentally-collinear points can score a spurious corr of 1.0. + min_overlap = max(_MIN_OVERLAP, len(era5_ws) // 2) + corrs = [_shift_corr(era5_ws=era5_ws, reference_ws=reference_ws, shift=s, min_overlap=min_overlap) for s in shifts] + sweep = pd.DataFrame({"shift_rows": shifts, "corr": corrs}) + if sweep["corr"].isna().all(): + return 0, float("nan"), sweep + best = sweep.loc[sweep["corr"].idxmax()] + return int(best["shift_rows"]), float(best["corr"]), sweep + + +def _shift_corr(*, era5_ws: pd.Series, reference_ws: pd.Series, shift: int, min_overlap: int) -> float: + """Pearson correlation of ``era5_ws.shift(shift)`` and ``reference_ws`` over finite pairs. + + Computed on the overlapping non-NaN pairs only, returning NaN below ``min_overlap`` points, + so extreme shifts neither raise numpy warnings (tests treat warnings as errors) nor score a + spurious perfect correlation off a handful of points. + """ + shifted = era5_ws.shift(shift) + pair = pd.concat([shifted, reference_ws], axis=1).dropna() + if len(pair) < min_overlap: + return float("nan") + a = pair.iloc[:, 0].to_numpy() + b = pair.iloc[:, 1].to_numpy() + if a.std() == 0 or b.std() == 0: # zero variance -> correlation undefined (and numpy warns) + return float("nan") + return float(np.corrcoef(a, b)[0, 1]) + + +def sync_era5( + era5_hourly_df: pd.DataFrame, + *, + target_index: pd.DatetimeIndex, + reference_ws: pd.Series, + timebase: pd.Timedelta | None = None, + max_lag: pd.Timedelta = _DEFAULT_MAX_LAG, +) -> Era5SyncResult: + """Upsample ERA5, find its lag vs ``reference_ws``, and align it to ``target_index``. + + :param era5_hourly_df: raw hourly ERA5 (Open-Meteo column names) + :param target_index: the analysis grid to align ERA5 onto (the SCADA timestamps) + :param reference_ws: a site wind speed on ``target_index`` (e.g. reference-turbine mean) + :param timebase: analysis timebase; inferred from ``target_index`` spacing when ``None`` + """ + if timebase is None: + timebase = pd.Timedelta(pd.Series(target_index).diff().median()) + upsampled = upsample_era5_to_timebase(era5_hourly_df, timebase=timebase).reindex(target_index) + best_lag, best_corr, sweep = find_best_lag( + reference_ws=reference_ws.reindex(target_index), + era5_ws=upsampled[ERA5_WS], + timebase=timebase, + max_lag=max_lag, + ) + aligned = upsampled.shift(best_lag) + return Era5SyncResult(aligned=aligned, best_lag_rows=best_lag, best_corr=best_corr, sweep=sweep) diff --git a/benchmarking/baselines/example_prepost_study.py b/benchmarking/baselines/example_prepost_study.py new file mode 100644 index 00000000..40d7a48e --- /dev/null +++ b/benchmarking/baselines/example_prepost_study.py @@ -0,0 +1,227 @@ +"""Driver: score v0 and the naive ratio method across all synthetic profiles on real Hill of Towie data. + +Wires the open Hill of Towie SCADA through the full Issue 3 stack in **prepost** mode: load -> +build the shared v0 context (metadata + ERA5) -> for each synthetic upgrade profile inject it, +build the replicate ensemble, run the **real** wind_up pre/post analysis per campaign via +:class:`V0BinnedMethod` alongside :class:`NaiveRatioMethod`, score against the injected truth, +and save a per-profile leaderboard CSV, the tidy per-replicate results, and a campaign-length +curve PNG. An ``oracle`` anchor is scored too, so its ~0 error confirms the harness is wired +correctly. + +This is the baseline every new method must beat (Issue 3 "Done when"). Prepost is where the +naive ratio method should struggle: pre and post periods do not share a wind distribution, so it +carries covariate-shift bias -- a useful contrast with its near-unbiased toggle behaviour (see +:mod:`benchmarking.baselines.example_toggle_study`). It is heavy: each ``(replicate, campaign)`` +is a full wind_up run, so the default sweep is a few dozen runs per profile. Reduce +``n_replicates`` / ``campaign_months`` for a quicker look. + +Run it:: + + uv run python -m benchmarking.baselines.example_prepost_study + +The first run downloads and caches the Hill of Towie v2 year zips from Zenodo and the ERA5 +reanalysis from Open-Meteo (needs the ``era5`` optional dependency group). Override the window, +output and cache directories via the ``main`` arguments or the ``WIND_UP_BENCHMARKING_*`` / +``WIND_UP_CACHE_DIR`` env vars. +""" + +from __future__ import annotations + +import logging +import os +from functools import partial +from pathlib import Path + +import matplotlib.pyplot as plt +import pandas as pd + +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.naive_ratio import NaiveRatioMethod +from benchmarking.baselines.power_model import CURATED_ERA5_EXCLUDE, TUNED_MODEL_PARAMS, PowerModelMethod +from benchmarking.baselines.v0_binned import V0BinnedMethod +from benchmarking.harness import Method, StudyConfig, leaderboard, plot_campaign_curves, score_study +from benchmarking.harness.example_hot_study import OracleMethod +from benchmarking.synthetic import HOT_COLUMNS, HOT_RATED_POWER_KW +from benchmarking.synthetic.make_example_datasets import example_profiles +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# The full stable, no-upgrade Hill of Towie window usable for this exercise: all of 2016-2020, +# well before the real T13 AeroUp (Sep 2021). Treatment start is drawn per replicate from +# 2018-01-01..2020-01-01. With min_pre_months=24 the earliest upgrade (2018-01-01) puts the +# baseline start at 2016-01-01 (the data start), giving v0 its full one-year detrend window even +# for the shortest campaign (whose detrend reaches ~21 months before the upgrade). Late upgrades +# near 2020 simply lose their longest campaign lengths (campaign_windows drops infeasible ones). +DEFAULT_START_DT = pd.Timestamp("2016-01-01", tz="UTC") +DEFAULT_END_DT_EXCL = pd.Timestamp("2021-01-01", tz="UTC") +DEFAULT_WTG_NUMBERS = [1, 3, 4, 7] +DEFAULT_TURBINE_SUBSET = [f"T{x:02d}" for x in DEFAULT_WTG_NUMBERS] +DEFAULT_TREATMENT_START_RANGE = (pd.Timestamp("2018-01-01", tz="UTC"), pd.Timestamp("2020-01-01", tz="UTC")) +MIN_PRE_MONTHS = 24 + + +def default_output_root() -> Path: + """Return the directory the prepost study writes its outputs under. + + Overridable via ``WIND_UP_BENCHMARKING_OUTPUT_DIR``; defaults to + ``~/temp/wind-up-benchmarking/prepost``. + """ + root = Path(os.getenv("WIND_UP_BENCHMARKING_OUTPUT_DIR", Path.home() / "temp" / "wind-up-benchmarking")) + return root / "prepost" + + +def save_per_method_curve(out_dir: Path, profile_name: str, method_name: str, method_results: pd.DataFrame) -> None: + """Write a single-method campaign-length curve the moment that method finishes. + + Same three panels as the combined :func:`plot_campaign_curves` plot (uplift recovery, bias +/- + spread, score) but for one method, so a long run can be sanity-checked method-by-method as each + completes -- catching a broken method early -- instead of only at the very end. Methods run + fastest-first (oracle, naive, then the slow wind_up run), so the cheap anchors appear first. + """ + summary = leaderboard(method_results) + fig = plot_campaign_curves( + summary, + save_path=out_dir / f"campaign_curves_{profile_name}_{method_name}.png", + title=f"{profile_name} - {method_name}", + ) + plt.close(fig) + logger.info("Saved per-method curve for %s (profile %s)", method_name, profile_name) + + +def run_prepost_study( + base_scada: pd.DataFrame, + *, + profiles: dict[str, list], + study: StudyConfig, + out_root: str | Path | None = None, + data_dir: str | Path | None = None, + include_oracle: bool = True, + include_v0: bool = True, +) -> pd.DataFrame: + """Score the methods (oracle anchor, naive, power_model, v0) over ``profiles`` and save outputs. + + :param base_scada: wind-up-format real SCADA (all subset turbines), the no-upgrade baseline + :param profiles: mapping of profile name -> list of upgrade callables to inject + :param study: the replicate/campaign sweep configuration (``mode="prepost"``) + :param out_root: output directory; defaults to :func:`default_output_root` + :param data_dir: Hill of Towie data/cache dir for the v0 context metadata; defaults to the + source package default (keep it the same as the dir ``base_scada`` was loaded from) + :param include_oracle: also score an oracle that returns the injected truth (sanity anchor) + :param include_v0: also score the v0 binned baseline; off-able because a real wind_up run per + campaign is very slow, so an initial oracle+naive+power_model pass is much quicker to review + :return: the concatenated tidy per-replicate results across all profiles + """ + out_dir = Path(out_root) if out_root is not None else default_output_root() + out_dir.mkdir(parents=True, exist_ok=True) + + context = build_hot_v0_context(data_dir=data_dir, wtg_names=DEFAULT_TURBINE_SUBSET) + scratch_dir = out_dir / "windup_runs" + + all_results = [] + for profile_name, profile in profiles.items(): + # Fastest first (oracle is instant, naive has no wind_up pipeline, v0 is a full wind_up run + # per campaign), so the per-method curves below appear early and a bad method is caught fast. + methods: list[Method] = [] + if include_oracle: + methods.append(OracleMethod(base_scada)) + methods.append( + NaiveRatioMethod( + columns=HOT_COLUMNS, + out_dir=out_dir / "naive_runs", + ) + ) + # The power model runs after the cheap naive floor (compare to it first) and before the slow + # v0 run. It is given curated reference-only features (each reference's active power + + # availability) plus all raw ERA5 columns (the context's hourly reanalysis frame). The HoT + # availability counter (wtc_ScReToOp_timeon) is wired as the test-turbine downtime filter + # (kept rows need a full timebase of availability, 600 s); the stuck-data filter and the + # finite-power rule also apply. + methods.append( + PowerModelMethod( + columns=HOT_COLUMNS, + baseline_rated_power_kw=HOT_RATED_POWER_KW, + era5_hourly_df=context.reanalysis_datasets[0].data, + # Removal-ablation accepted defaults (findings F13). + availability_feature=False, + era5_exclude=CURATED_ERA5_EXCLUDE, + # Issue 12 accepted default (findings F14): looser leaf capacity. + model_params=dict(TUNED_MODEL_PARAMS), + out_dir=out_dir / "power_model_runs", + ) + ) + if include_v0: + methods.append(V0BinnedMethod(context, scratch_dir=scratch_dir)) + logger.info("Scoring prepost profile %s with methods %s", profile_name, [m.name for m in methods]) + results = score_study( + base_scada, + profile=profile, + methods=methods, + study=study, + profile_name=profile_name, + on_method_complete=partial(save_per_method_curve, out_dir, profile_name), + ) + summary = leaderboard(results) + + results.to_csv(out_dir / f"results_{profile_name}.csv", index=False) + summary.to_csv(out_dir / f"leaderboard_{profile_name}.csv", index=False) + plot_campaign_curves(summary, save_path=out_dir / f"campaign_curves_{profile_name}.png", title=profile_name) + all_results.append(results) + + combined = pd.concat(all_results, ignore_index=True) + combined_summary = leaderboard(combined) + combined_summary.to_csv(out_dir / "leaderboard_all_profiles.csv", index=False) + logger.info("Prepost leaderboard (all profiles):\n%s", combined_summary.to_string(index=False)) + return combined + + +def main( + *, + out_root: str | Path | None = None, + data_dir: str | Path | None = None, + start_dt: pd.Timestamp = DEFAULT_START_DT, + end_dt_excl: pd.Timestamp = DEFAULT_END_DT_EXCL, + wtg_numbers: list[int] | None = None, + n_replicates: int = 4, + campaign_months: list[int] | None = None, + include_v0: bool = True, +) -> pd.DataFrame: + """Run the prepost study end-to-end on real Hill of Towie data and save outputs. + + :param out_root: output directory; defaults to :func:`default_output_root` + :param data_dir: Hill of Towie data/cache dir; defaults to the source package default + :param start_dt: inclusive UTC window start + :param end_dt_excl: exclusive UTC window end + :param wtg_numbers: turbine numbers to load; defaults to the stable south-west cluster + :param n_replicates: ensemble size per profile (each replicate x campaign is a full v0 run) + :param campaign_months: the campaign-length sweep grid, in months + :param include_v0: also score the slow v0 baseline (set False for a quick oracle+naive+power_model pass) + :return: the combined tidy results across all profiles + """ + wtg_numbers = wtg_numbers if wtg_numbers is not None else DEFAULT_WTG_NUMBERS + campaign_months = campaign_months if campaign_months is not None else [3, 6, 9, 12] + logger.info("Loading Hill of Towie SCADA %s..%s for turbines %s", start_dt, end_dt_excl, wtg_numbers) + scada_df, _metadata_df = load_hot_scada( + start_dt=start_dt, + end_dt_excl=end_dt_excl, + wtg_numbers=wtg_numbers, + wtg_names=DEFAULT_TURBINE_SUBSET, + data_dir=Path(data_dir) if data_dir is not None else None, + ) + study = StudyConfig( + mode="prepost", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=campaign_months, + n_replicates=n_replicates, + seed=0, + ) + return run_prepost_study( + scada_df, profiles=example_profiles(), study=study, out_root=out_root, data_dir=data_dir, include_v0=include_v0 + ) + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") + main() diff --git a/benchmarking/baselines/example_toggle_study.py b/benchmarking/baselines/example_toggle_study.py new file mode 100644 index 00000000..599d446d --- /dev/null +++ b/benchmarking/baselines/example_toggle_study.py @@ -0,0 +1,198 @@ +"""Driver: score v0 and the naive ratio method on a TOGGLE campaign on real Hill of Towie data. + +Mirrors :mod:`benchmarking.baselines.example_prepost_study` but runs in toggle mode: a fast +20-min-on / 20-min-off schedule (``toggle_period = 40min``) and a single 3% constant-Cp upgrade. +``V0BinnedMethod`` runs wind_up's native toggle assessment; ``NaiveRatioMethod`` splits on/off +via the shared toggle mask. An oracle anchor is scored too, so its ~0 error confirms the +toggle harness path is wired correctly. + +Toggle is where the naive ratio method should shine: interleaved on/off blocks share a wind +distribution, so it carries little covariate-shift bias -- a useful contrast with its prepost +behaviour in the v0 study. + +Run it:: + + uv run python -m benchmarking.baselines.example_toggle_study + +The first run downloads and caches the Hill of Towie v2 year zips from Zenodo and the ERA5 +reanalysis from Open-Meteo (needs the ``era5`` optional dependency group). +""" + +from __future__ import annotations + +import logging +import os +from functools import partial +from pathlib import Path + +import pandas as pd + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TREATMENT_START_RANGE, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, + MIN_PRE_MONTHS, + save_per_method_curve, +) +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.naive_ratio import NaiveRatioMethod +from benchmarking.baselines.power_model import CURATED_ERA5_EXCLUDE, TUNED_MODEL_PARAMS, PowerModelMethod +from benchmarking.baselines.v0_binned import V0BinnedMethod +from benchmarking.harness import Method, StudyConfig, leaderboard, plot_campaign_curves, score_study +from benchmarking.harness.example_hot_study import OracleMethod +from benchmarking.synthetic import HOT_COLUMNS, HOT_RATED_POWER_KW, ConstantCpChange +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# 20 minutes on, 20 minutes off -> a 40-minute on/off cycle. +DEFAULT_TOGGLE_PERIOD = pd.Timedelta(minutes=40) +# One upgrade for the toggle study: a flat +3% Cp change in region 2. +TOGGLE_PROFILES: dict[str, list] = {"cp_plus_3pct": [ConstantCpChange(delta=0.03)]} + + +def default_output_root() -> Path: + """Return the directory the toggle study writes its outputs under. + + Overridable via ``WIND_UP_BENCHMARKING_OUTPUT_DIR``; defaults to + ``~/temp/wind-up-benchmarking/toggle``. + """ + root = Path(os.getenv("WIND_UP_BENCHMARKING_OUTPUT_DIR", Path.home() / "temp" / "wind-up-benchmarking")) + return root / "toggle" + + +def run_toggle_study( + base_scada: pd.DataFrame, + *, + profiles: dict[str, list], + study: StudyConfig, + out_root: str | Path | None = None, + data_dir: str | Path | None = None, + include_oracle: bool = True, + include_v0: bool = True, +) -> pd.DataFrame: + """Score the methods (oracle anchor, naive, power_model, v0) on a toggle study and save outputs. + + :param base_scada: wind-up-format real SCADA (all subset turbines), the no-upgrade baseline + :param profiles: mapping of profile name -> list of upgrade callables to inject + :param study: the replicate/campaign sweep configuration (``mode="toggle"``) + :param out_root: output directory; defaults to :func:`default_output_root` + :param data_dir: Hill of Towie data/cache dir for the v0 context metadata; defaults to the + source package default (keep it the same as the dir ``base_scada`` was loaded from) + :param include_oracle: also score an oracle that returns the injected truth (sanity anchor) + :param include_v0: also score the slow v0 baseline (set False for a quick oracle+naive+power_model pass) + :return: the concatenated tidy per-replicate results across all profiles + """ + out_dir = Path(out_root) if out_root is not None else default_output_root() + out_dir.mkdir(parents=True, exist_ok=True) + + context = build_hot_v0_context(data_dir=data_dir, wtg_names=DEFAULT_TURBINE_SUBSET) + scratch_dir = out_dir / "windup_runs" + + all_results = [] + for profile_name, profile in profiles.items(): + # Fastest first (oracle is instant, naive has no wind_up pipeline, v0 is a full wind_up run + # per campaign), so the per-method curves appear early and a bad method is caught fast. + methods: list[Method] = [] + if include_oracle: + methods.append(OracleMethod(base_scada)) + methods.append( + NaiveRatioMethod( + columns=HOT_COLUMNS, + out_dir=out_dir / "naive_runs", + ) + ) + # Power model after the naive floor, before slow v0; curated reference-only features + ERA5. + methods.append( + PowerModelMethod( + columns=HOT_COLUMNS, + baseline_rated_power_kw=HOT_RATED_POWER_KW, + era5_hourly_df=context.reanalysis_datasets[0].data, + # Removal-ablation accepted defaults (findings F13). + availability_feature=False, + era5_exclude=CURATED_ERA5_EXCLUDE, + # Issue 12 accepted default (findings F14): looser leaf capacity. + model_params=dict(TUNED_MODEL_PARAMS), + out_dir=out_dir / "power_model_runs", + ) + ) + if include_v0: + methods.append(V0BinnedMethod(context, scratch_dir=scratch_dir)) + logger.info("Scoring toggle profile %s with methods %s", profile_name, [m.name for m in methods]) + results = score_study( + base_scada, + profile=profile, + methods=methods, + study=study, + profile_name=profile_name, + on_method_complete=partial(save_per_method_curve, out_dir, profile_name), + ) + summary = leaderboard(results) + + results.to_csv(out_dir / f"results_{profile_name}.csv", index=False) + summary.to_csv(out_dir / f"leaderboard_{profile_name}.csv", index=False) + plot_campaign_curves(summary, save_path=out_dir / f"campaign_curves_{profile_name}.png", title=profile_name) + all_results.append(results) + + combined = pd.concat(all_results, ignore_index=True) + combined_summary = leaderboard(combined) + combined_summary.to_csv(out_dir / "leaderboard_all_profiles.csv", index=False) + logger.info("Toggle leaderboard (all profiles):\n%s", combined_summary.to_string(index=False)) + return combined + + +def main( + *, + out_root: str | Path | None = None, + data_dir: str | Path | None = None, + start_dt: pd.Timestamp = DEFAULT_START_DT, + end_dt_excl: pd.Timestamp = DEFAULT_END_DT_EXCL, + wtg_numbers: list[int] | None = None, + n_replicates: int = 4, + campaign_months: list[int] | None = None, + toggle_period: pd.Timedelta = DEFAULT_TOGGLE_PERIOD, + include_v0: bool = True, +) -> pd.DataFrame: + """Run the toggle study end-to-end on real Hill of Towie data and save outputs. + + :param out_root: output directory; defaults to :func:`default_output_root` + :param data_dir: Hill of Towie data/cache dir; defaults to the source package default + :param start_dt: inclusive UTC window start + :param end_dt_excl: exclusive UTC window end + :param wtg_numbers: turbine numbers to load; defaults to the stable south-west cluster + :param n_replicates: ensemble size per profile + :param campaign_months: the campaign-length (toggling-duration) sweep grid, in months + :param toggle_period: the on/off cycle length (20 min on + 20 min off = 40 min by default) + :param include_v0: also score the slow v0 baseline (set False for a quick oracle+naive+power_model pass) + :return: the combined tidy results across all profiles + """ + wtg_numbers = wtg_numbers if wtg_numbers is not None else DEFAULT_WTG_NUMBERS + campaign_months = campaign_months if campaign_months is not None else [3, 6, 9, 12] + logger.info("Loading Hill of Towie SCADA %s..%s for turbines %s", start_dt, end_dt_excl, wtg_numbers) + scada_df, _metadata_df = load_hot_scada( + start_dt=start_dt, + end_dt_excl=end_dt_excl, + wtg_numbers=wtg_numbers, + wtg_names=DEFAULT_TURBINE_SUBSET, + data_dir=Path(data_dir) if data_dir is not None else None, + ) + study = StudyConfig( + mode="toggle", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=campaign_months, + toggle_period=toggle_period, + n_replicates=n_replicates, + seed=0, + ) + return run_toggle_study( + scada_df, profiles=TOGGLE_PROFILES, study=study, out_root=out_root, data_dir=data_dir, include_v0=include_v0 + ) + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") + main() diff --git a/benchmarking/baselines/filtering.py b/benchmarking/baselines/filtering.py new file mode 100644 index 00000000..444b0252 --- /dev/null +++ b/benchmarking/baselines/filtering.py @@ -0,0 +1,83 @@ +"""Shared test-turbine normal-operation filtering for the benchmarking methods. + +The outcome is the test turbine's power, so abnormal operation unrelated to the upgrade — +downtime, curtailment, frozen/stuck sensors — would otherwise be attributed to the upgrade (a +downward bias, worst when it clusters in the upgrade period). Every method must select the +normally-operating test-turbine rows the same way, so this filter lives in one shared place +(the R-learner and the naive ratio both use it; it has no ``wind_up`` dependency). + +Three checks: + +* **finite power** — rows with NaN active power are downtime / missing energy, always dropped. +* **downtime / availability** — drop rows where an availability counter shows the turbine was not + ready to operate for the full period. This is **required** by the methods (a missing availability + column is a configuration error, not a silent no-op). +* **stuck data** — drop rows where every signal is unchanged from the previous record (a frozen + data stream), exempting genuine very-low-wind calms. + +The central rule is **filter on cause, not effect**: selection uses operational signals and finite +power, never "power lower than expected" — that would drop genuine low-uplift records and bias the +estimate. This is row selection, not a feature rule, so using the test turbine's own operational +signals here does not violate the upgrade-invariant feature rule. References are deliberately not +filtered (the R-learner learns their operating modes; the naive ratio keeps complete-case refs). +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + import pandas as pd + +# Below this wind speed a flat/constant signal is a genuine calm, not a stuck sensor. +_VERY_LOW_WIND = 1.5 + + +@dataclass +class NormalOperationFilter: + """Selects normally-operating test-turbine timestamps (cause, not effect). + + :param active_power_col: the test turbine's active-power column (rows with NaN power are + always dropped — that is downtime/missing energy) + :param wind_speed_col: the test turbine's wind-speed column, used only to exempt very-low-wind + calms from the stuck filter (``None`` disables that exemption) + :param availability_col: an operational "ready to operate" counter (e.g. seconds in the + period); ``None`` disables the downtime filter + :param full_period_seconds: the counter value that means fully available; defaults to the + timebase length in seconds when ``None`` + :param apply_stuck_filter: drop frozen/stuck rows (all signals unchanged vs the previous row) + """ + + active_power_col: str + wind_speed_col: str | None = None + availability_col: str | None = None + full_period_seconds: float | None = None + apply_stuck_filter: bool = True + + def keep_mask(self, test_rows: pd.DataFrame, *, timebase: pd.Timedelta) -> pd.Series: + """Boolean Series (True = keep) of normally-operating test-turbine rows, index-aligned.""" + rows = test_rows.sort_index() + keep = rows[self.active_power_col].notna() + if self.apply_stuck_filter: + keep &= ~self._stuck(rows) + if self.availability_col is not None: + keep &= self._available(rows, timebase=timebase) + return keep.astype(bool) + + def _stuck(self, rows: pd.DataFrame) -> pd.Series: + """Return True where every numeric signal is unchanged from the previous row (not low wind).""" + numeric = rows.select_dtypes(include="number") + diffs = numeric.ffill().fillna(0).diff() + frozen = (diffs == 0).all(axis=1) + frozen.iloc[0] = False # the first row has no predecessor to repeat + if self.wind_speed_col is not None: + calm = rows[self.wind_speed_col] < _VERY_LOW_WIND + frozen &= ~calm + return frozen + + def _available(self, rows: pd.DataFrame, *, timebase: pd.Timedelta) -> pd.Series: + """Return True where the availability counter shows a full period (NaN -> not available).""" + full = self.full_period_seconds if self.full_period_seconds is not None else timebase.total_seconds() + counter = rows[self.availability_col] + return (counter >= full) & counter.notna() diff --git a/benchmarking/baselines/hot_context.py b/benchmarking/baselines/hot_context.py new file mode 100644 index 00000000..7f6c069f --- /dev/null +++ b/benchmarking/baselines/hot_context.py @@ -0,0 +1,71 @@ +"""Shared Hill of Towie source-context for v0-style assessment methods. + +``build_hot_v0_context`` assembles the source-specific inputs a wind_up assessment needs but +the harness's thin ``MethodInput`` does not carry: per-turbine metadata (lat/long), ERA5 +reanalysis, and the paths to the vendored asset and northing-corrections YAMLs. It is loaded +once and reused across every campaign a method scores. + +This is a benchmarking helper, not a harness-enforced contract: future methods (e.g. the +Issue 5 R-learner) reuse it by calling it from their own constructor. Keeping it here lets the +harness seam stay thin until the Issue 4 contract has two real consumers to design against. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import TYPE_CHECKING + +from benchmarking.synthetic.sources.hill_of_towie import load_hot_metadata +from wind_up.era5 import get_era5_hourly_df +from wind_up.reanalysis_data import ReanalysisDataset + +if TYPE_CHECKING: + from collections.abc import Sequence + + import pandas as pd + +ASSETS_DIR = Path(__file__).parent / "assets" +ASSET_YAML = ASSETS_DIR / "HOT.yaml" +NORTHING_YAML = ASSETS_DIR / "optimized_northing_corrections.yaml" + +HOT_LAT: float = 57.50 +HOT_LON: float = -3.25 +HOT_ERA5_START: str = "2000-01-01" +HOT_ERA5_END: str = "2026-05-01" + + +def get_hot_reanalysis_datasets() -> list[ReanalysisDataset]: + """Return a list with one :class:`ReanalysisDataset` for the Hill of Towie site.""" + return [ + ReanalysisDataset( + id=f"ERA5_{HOT_LAT:.2f}_{HOT_LON:.2f}", + data=get_era5_hourly_df(lat=HOT_LAT, lon=HOT_LON, start_date=HOT_ERA5_START, end_date=HOT_ERA5_END), + ) + ] + + +@dataclass +class HotV0Context: + """Source-specific inputs a v0-style assessment needs, loaded once and reused. + + :param metadata_df: per-turbine metadata (Name, Latitude, Longitude) in wind-up format + :param reanalysis_datasets: ERA5 reanalysis datasets for the HoT site + :param asset_yaml: path to the vendored asset YAML (turbine list + type) + :param northing_yaml: path to the vendored optimized northing-corrections YAML + """ + + metadata_df: pd.DataFrame + reanalysis_datasets: list[ReanalysisDataset] + asset_yaml: Path = ASSET_YAML + northing_yaml: Path = NORTHING_YAML + + +def build_hot_v0_context(*, data_dir: str | Path | None = None, wtg_names: Sequence[str] | None = None) -> HotV0Context: + """Assemble the HoT v0 context: load metadata and (fetch+cache) ERA5 reanalysis once. + + :param data_dir: Hill of Towie data/cache dir; defaults to the source package default + """ + metadata_df = load_hot_metadata(data_dir=Path(data_dir) if data_dir is not None else None, wtg_names=wtg_names) + reanalysis_datasets = get_hot_reanalysis_datasets() + return HotV0Context(metadata_df=metadata_df, reanalysis_datasets=reanalysis_datasets) diff --git a/benchmarking/baselines/inspect_era5_matching_importance.py b/benchmarking/baselines/inspect_era5_matching_importance.py new file mode 100644 index 00000000..e3125186 --- /dev/null +++ b/benchmarking/baselines/inspect_era5_matching_importance.py @@ -0,0 +1,311 @@ +"""One-off analysis: which ERA5 fields best predict the test turbine's power (Issue 8, Component 1). + +The bias-cancellation correction (Issue 8) matches the baseline and upgraded periods on **ERA5-only** +weather so a common per-bin multiplicative shrinkage cancels between the two train/predict directions +(design + ``docs/v1/findings.md`` F5). CEM cell count explodes with dimension, so we can only afford to +match on a *few* ERA5 variables — and they must be the ones that actually drive the test turbine's +power, or the matched shrinkage factor is not the one that distorts the estimate. + +This script ranks the ERA5 fields by how well they predict the test turbine's (un-upgraded, real HoT) +power, using two views so gain alone is not over-trusted: + +* **LightGBM gain importance** from the same outcome-model factory the method uses; +* **sklearn permutation importance** on a held-out split (model-agnostic, guards against gain quirks). + +ERA5-only (not the full reference+ERA5 matrix) because the reference active-power features otherwise +dominate and mask the ERA5 signal, and only ERA5 has the full-coverage, temporally-stable columns we +can actually match on. The whole default window is genuine no-upgrade HoT SCADA, so every +normally-operating row is a "baseline" row for this purpose. + +Outputs (under ``/inspection_era5_matching``): + +* ``feature_importance.png`` — gain and permutation rankings side by side (the selection view); +* ``predicted_vs_actual.png`` — held-out predicted vs actual test power with R²/RMSE/MAE. This is a + **gate**: if ERA5 predicts test power poorly the whole ranking is suspect, not just imprecise. +* ``era5_matching_importance.csv`` — the merged ranking table. + +The chosen matching set + rationale (citing these metrics) is recorded as **F6** in +``docs/v1/findings.md`` and hard-coded as the method default; this script does not edit anything. + +Run from the repo root:: + + uv run python -m benchmarking.baselines.inspect_era5_matching_importance + uv run python -m benchmarking.baselines.inspect_era5_matching_importance --test-wtg T07 +""" + +from __future__ import annotations + +import argparse +import logging +from dataclasses import dataclass +from pathlib import Path + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +from sklearn.inspection import permutation_importance + +from benchmarking.baselines.era5_derived import shear_exponent +from benchmarking.baselines.era5_sync import sync_era5 +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, + default_output_root, +) +from benchmarking.baselines.filtering import NormalOperationFilter +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.power_model.features import ( + era5_feature_frame, + extract_outcome, + reference_mean_wind_speed, +) +from benchmarking.baselines.rlearner.nuisance import make_outcome_model +from benchmarking.diagnostics.density import density_scatter +from benchmarking.diagnostics.style import apply_grid, save_fig +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# Prefer a small matching set; flag the strongest few above a fraction-of-the-top-feature floor. +_SELECTION_FLOOR_FRAC = 0.05 +_PREFERRED_N = 2 + + +def _infer_timebase(index: pd.DatetimeIndex) -> pd.Timedelta: + """Analysis timebase as the median spacing of the sorted unique timestamps (≈10 min for HoT).""" + unique = pd.DatetimeIndex(pd.unique(index)).sort_values() + if len(unique) < 2: # noqa: PLR2004 + return pd.Timedelta(minutes=10) + return pd.Timedelta(np.median(np.diff(unique.to_numpy()))) + + +def add_shear_exponent(features: pd.DataFrame) -> pd.DataFrame: + """Fold the collinear 10m/100m wind speeds into one physical vertical-shear exponent. + + The two ERA5 wind speeds are strongly correlated, so gain splits credit between them while + permutation discounts whichever is redundant — neither view then cleanly reflects the *shear* they + jointly encode. The power-law exponent (see + :func:`benchmarking.baselines.era5_derived.shear_exponent`, the shared Issue 9 utility) captures + that shear in a single column (a stability / turbulence proxy that directly attacks the F5 cause), + so we keep ``wind_speed_100m`` as the magnitude and drop the now-redundant ``wind_speed_10m``. + """ + alpha = shear_exponent(features["wind_speed_10m"], features["wind_speed_100m"]) + return features.assign(wind_shear_exponent=alpha.to_numpy()).drop(columns=["wind_speed_10m"]) + + +def build_era5_and_outcome(scada_df: pd.DataFrame, *, test_wtg: str) -> tuple[pd.DataFrame, pd.Series]: + """Return (ERA5-only feature frame, test-turbine power) over normally-operating, finite rows. + + ERA5 is synced to the SCADA grid via the reference-turbine mean wind speed (upgrade-invariant, the + same signal the method's lag sweep locks onto). No synthetic upgrade is injected, so the whole + window is baseline; the normal-operation filter drops curtailed/down rows that would otherwise + corrupt the target. The 10m/100m speeds are folded into a vertical-shear exponent + (:func:`add_shear_exponent`). + """ + index = pd.DatetimeIndex(pd.unique(scada_df.index)).sort_values() + timebase = _infer_timebase(index) + context = build_hot_v0_context(wtg_names=list(scada_df[HOT_COLUMNS.turbine].unique())) + + y = extract_outcome( + scada_df, test_wtg=test_wtg, turbine_col=HOT_COLUMNS.turbine, active_power_col=HOT_COLUMNS.active_power + ) + reference_ws = reference_mean_wind_speed( + scada_df, test_wtg=test_wtg, turbine_col=HOT_COLUMNS.turbine, wind_speed_col=HOT_COLUMNS.wind_speed + ) + synced = sync_era5( + context.reanalysis_datasets[0].data, target_index=index, reference_ws=reference_ws, timebase=timebase + ) + raw = era5_feature_frame(synced.aligned) + logger.info("ERA5 synced: lag=%d rows, corr=%.3f", synced.best_lag_rows, synced.best_corr) + + test_rows = scada_df[scada_df[HOT_COLUMNS.turbine] == test_wtg].sort_index() + keep = NormalOperationFilter( + active_power_col=HOT_COLUMNS.active_power, + wind_speed_col=HOT_COLUMNS.wind_speed, + availability_col=HOT_COLUMNS.availability, + ).keep_mask(test_rows, timebase=timebase) + keep = keep[~keep.index.duplicated()].reindex(index, fill_value=False).to_numpy() + # Finite-check on the raw ERA5 columns (all present in reanalysis) so the derived shear NaN on rare + # calm rows does not shrink the row set — keeping this comparable to the pre-shear run. + selected = keep & np.isfinite(y.to_numpy(dtype=float)) & raw.notna().all(axis=1).to_numpy() + + features = add_shear_exponent(raw) + return features.loc[selected], y.loc[selected] + + +@dataclass +class RankingResult: + """The ERA5 importance ranking plus the held-out slice it was scored on. + + :param table: one row per ERA5 feature with gain / gain_frac / permutation importance, gain-sorted + :param y_valid: actual test power on the held-out slice (feeds the predicted-vs-actual gate) + :param pred_valid: model prediction on the held-out slice + :param n_train: rows the ranking model was fitted on + """ + + table: pd.DataFrame + y_valid: np.ndarray + pred_valid: np.ndarray + n_train: int + + +def rank_features(features: pd.DataFrame, y: pd.Series, *, seed: int) -> RankingResult: + """Rank ERA5 features by LightGBM gain + held-out permutation importance on one seeded split. + + Fits on a seeded 80% split; gain comes from the fitted booster, permutation importance and the + held-out predictions (returned for the predicted-vs-actual gate) both come off the untouched 20% + so nothing is scored in-sample. + """ + rng = np.random.default_rng(seed) + order = rng.permutation(len(y)) + n_valid = max(1, len(y) // 5) + valid_idx, train_idx = order[:n_valid], order[n_valid:] + x_train, x_valid = features.iloc[train_idx], features.iloc[valid_idx] + y_train, y_valid = y.to_numpy(dtype=float)[train_idx], y.to_numpy(dtype=float)[valid_idx] + + model = make_outcome_model(random_state=seed) + model.fit(x_train, y_train) + gain = model.booster_.feature_importance(importance_type="gain").astype(float) + perm = permutation_importance(model, x_valid, y_valid, n_repeats=10, random_state=seed, scoring="r2") + + table = pd.DataFrame( + { + "feature": list(features.columns), + "gain": gain, + "gain_frac": gain / gain.sum() if gain.sum() else np.nan, + "perm_importance": perm.importances_mean, + "perm_importance_std": perm.importances_std, + } + ).sort_values("gain", ascending=False, ignore_index=True) + + pred_valid = np.asarray(model.predict(x_valid), dtype=float) + return RankingResult(table=table, y_valid=y_valid, pred_valid=pred_valid, n_train=len(y_train)) + + +def _fit_metrics(actual: np.ndarray, predicted: np.ndarray) -> dict[str, float]: + """R², RMSE, MAE over the finite pairs (the predicted-vs-actual gate numbers).""" + finite = np.isfinite(actual) & np.isfinite(predicted) + actual, predicted = actual[finite], predicted[finite] + resid = actual - predicted + ss_tot = float(np.sum((actual - actual.mean()) ** 2)) + r2 = 1.0 - float(np.sum(resid**2)) / ss_tot if ss_tot else float("nan") + return {"r2": r2, "rmse": float(np.sqrt(np.mean(resid**2))), "mae": float(np.mean(np.abs(resid)))} + + +def _select_matching_vars(table: pd.DataFrame) -> list[str]: + """Suggested matching set: features above a fraction-of-top gain floor, capped at the preferred count.""" + if table.empty: + return [] + floor = _SELECTION_FLOOR_FRAC * float(table["gain"].iloc[0]) + above = table[table["gain"] >= floor]["feature"].tolist() + return above[:_PREFERRED_N] + + +def plot_feature_importance(table: pd.DataFrame, *, test_wtg: str, out_dir: Path, top_n: int = 20) -> None: + """Gain and permutation-importance rankings side by side — the matching-variable selection view.""" + top = table.head(top_n).iloc[::-1] + fig, axes = plt.subplots(1, 2, figsize=(15, max(6.0, 0.4 * len(top)))) + axes[0].barh(top["feature"], top["gain"], color="C0") + axes[0].set_xlabel("LightGBM gain") + axes[0].set_title("gain importance") + apply_grid(axes[0]) + perm_order = table.head(top_n).sort_values("perm_importance") + axes[1].barh( + perm_order["feature"], perm_order["perm_importance"], xerr=perm_order["perm_importance_std"], color="C1" + ) + axes[1].set_xlabel("permutation importance (Δ R², held-out)") + axes[1].set_title("permutation importance") + apply_grid(axes[1]) + fig.suptitle(f"{test_wtg}: ERA5 → test power importance — match on the strongest, cheapest-to-match few") + save_fig(fig, out_dir / "feature_importance.png") + + +def plot_predicted_vs_actual(y_valid: np.ndarray, pred_valid: np.ndarray, *, test_wtg: str, out_dir: Path) -> None: + """Held-out predicted vs actual test power — the gate that the ERA5 model is good enough to trust.""" + metrics = _fit_metrics(y_valid, pred_valid) + fig, ax = plt.subplots(figsize=(7.5, 7)) + density_scatter(y_valid, pred_valid, ax=ax, s=6, colorbar=True) + hi = float(np.nanmax(y_valid)) if len(y_valid) else 1.0 + ax.plot([0.0, hi], [0.0, hi], color="red", linewidth=1.2, label="1:1") + ax.set_xlabel("actual test power [kW]") + ax.set_ylabel("predicted test power [kW]") + ax.set_title( + f"{test_wtg}: ERA5-only held-out fit " + f"R²={metrics['r2']:.3f}, RMSE={metrics['rmse']:.0f} kW, MAE={metrics['mae']:.0f} kW, n={len(y_valid)}" + ) + ax.legend(loc="upper left") + apply_grid(ax) + save_fig(fig, out_dir / "predicted_vs_actual.png") + + +def run(*, test_wtg: str, out_root: Path | None, seed: int) -> pd.DataFrame: + """Load HoT + ERA5, rank the ERA5 fields for ``test_wtg``, and write the plots + ranking CSV.""" + out_dir = (out_root if out_root is not None else default_output_root()) / "inspection_era5_matching" + out_dir.mkdir(parents=True, exist_ok=True) + + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + features, y = build_era5_and_outcome(scada_df, test_wtg=test_wtg) + logger.info("Ranking %d ERA5 features on %d normally-operating rows for %s", features.shape[1], len(y), test_wtg) + + result = rank_features(features, y, seed=seed) + table = result.table + table.to_csv(out_dir / "era5_matching_importance.csv", index=False) + plot_feature_importance(table, test_wtg=test_wtg, out_dir=out_dir) + plot_predicted_vs_actual(result.y_valid, result.pred_valid, test_wtg=test_wtg, out_dir=out_dir) + + metrics = _fit_metrics(result.y_valid, result.pred_valid) + logger.info( + "Held-out ERA5→test-power fit: R²=%.3f, RMSE=%.0f kW, MAE=%.0f kW (n_train=%d, n_valid=%d)", + metrics["r2"], + metrics["rmse"], + metrics["mae"], + result.n_train, + len(result.y_valid), + ) + logger.info( + "ERA5 feature ranking (top 12 by gain):\n%s", + table.head(12)[["feature", "gain", "gain_frac", "perm_importance"]].round(4).to_string(index=False), + ) + logger.info( + "Suggested matching set (gain >= %.0f%% of top, capped at %d): %s — verify against the cell budget " + "(Component 2) before hard-coding as the F6 default.", + 100 * _SELECTION_FLOOR_FRAC, + _PREFERRED_N, + _select_matching_vars(table), + ) + logger.info("Wrote ERA5 matching-importance outputs to %s", out_dir) + return table + + +def main() -> None: + """CLI: rank the ERA5 fields for one test turbine and write the analysis outputs.""" + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument( + "--test-wtg", + default=DEFAULT_TURBINE_SUBSET[0], + choices=DEFAULT_TURBINE_SUBSET, + help="test turbine whose power is the prediction target (default: the first study turbine)", + ) + parser.add_argument( + "--out-root", + type=Path, + default=None, + help="base output dir; the run writes under /inspection_era5_matching " + "(default: the study output root)", + ) + parser.add_argument("--seed", type=int, default=0, help="seed for the train/valid split and the model") + args = parser.parse_args() + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s", force=True) + run(test_wtg=args.test_wtg, out_root=args.out_root.expanduser() if args.out_root else None, seed=args.seed) + + +if __name__ == "__main__": + main() diff --git a/benchmarking/baselines/inspect_naive.py b/benchmarking/baselines/inspect_naive.py new file mode 100644 index 00000000..fd39633d --- /dev/null +++ b/benchmarking/baselines/inspect_naive.py @@ -0,0 +1,189 @@ +"""Manual inspection driver: run a few naive-ratio replicates (prepost and toggle) with plots on. + +A runnable companion to :mod:`benchmarking.baselines.inspect_v0_run`, but for +:class:`benchmarking.baselines.naive_ratio.NaiveRatioMethod`. It runs a handful of replicates in +each mode with ``save_plots=True``, each in its **own output directory**, so the naive method's +diagnostics (the per-run data-stats / results CSVs and the scatter, ratio-timeseries and +used-data-coverage plots) can +be eyeballed to confirm it received and interpreted the data correctly. The naive method has no +wind_up dependency, so no v0 context / metadata is needed. + +Each replicate is built and scored exactly as :func:`benchmarking.harness.score_study` would +(same windowing and ground truth), so what you inspect is the real scored path. + +Run it:: + + uv run python -m benchmarking.baselines.inspect_naive + +First run downloads and caches the Hill of Towie v2 SCADA (Zenodo). Each replicate's outputs land +under ``/inspection_naive//replicate__/naive__/`` (with a +``plots`` subfolder). +""" + +from __future__ import annotations + +import logging +from pathlib import Path + +import pandas as pd + +from benchmarking.baselines.example_prepost_study import default_output_root +from benchmarking.baselines.naive_ratio import NaiveRatioMethod +from benchmarking.harness import ( + MethodInput, + StudyConfig, + build_replicates, + campaign_windows, + treated_activity_mask, + window_row_mask, +) +from benchmarking.synthetic import HOT_COLUMNS, ConstantCpChange +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# Same 2016-2020 stable window as inspect_v0_run: treatment drawn per replicate from +# 2018-01-01..2019-12-31, with a 24-month baseline. +DEFAULT_START_DT = pd.Timestamp("2016-01-01", tz="UTC") +DEFAULT_END_DT_EXCL = pd.Timestamp("2021-01-01", tz="UTC") +DEFAULT_WTG_NUMBERS = [1, 3, 4, 7] +DEFAULT_TURBINE_SUBSET = [f"T{x:02d}" for x in DEFAULT_WTG_NUMBERS] +DEFAULT_TREATMENT_START_RANGE = (pd.Timestamp("2018-01-01", tz="UTC"), pd.Timestamp("2019-12-31 23:50", tz="UTC")) +MIN_PRE_MONTHS = 24 +# 20 minutes on, 20 minutes off -> a 40-minute on/off cycle. +DEFAULT_TOGGLE_PERIOD = pd.Timedelta(minutes=40) + + +def _inspect_mode(scada_df: pd.DataFrame, *, study: StudyConfig, out_dir: Path, delta: float) -> list[dict]: + """Run every replicate of ``study`` with the naive method (plots on) and return summary rows.""" + out_dir.mkdir(parents=True, exist_ok=True) + replicates = build_replicates(scada_df, profile=[ConstantCpChange(delta=delta)], study=study) + + rows = [] + for rep in replicates: + windows = campaign_windows( + rep.treatment_start, + min_pre_months=study.min_pre_months, + campaign_months=study.campaign_months, + data_start=scada_df.index.min(), + data_end=scada_df.index.max(), + ) + if not windows: + logger.warning("replicate %d (%s): no feasible campaign window, skipping", rep.replicate_id, rep.test_wtg) + continue + window = windows[-1] # the longest campaign -> most data to eyeball + + syn = rep.synthetic_df + mi = MethodInput( + scada_df=syn.loc[window_row_mask(syn.index, window)], + test_wtg=rep.test_wtg, + upgrade_timing=rep.upgrade_timing, + turbine_col=HOT_COLUMNS.turbine, + ) + rep_dir = out_dir / f"replicate_{rep.replicate_id:02d}_{rep.test_wtg}" + estimate = ( + NaiveRatioMethod( + columns=HOT_COLUMNS, + out_dir=rep_dir, + save_plots=True, + ) + .estimate(mi) + .p50_overall + ) + + test_index = syn.loc[syn[HOT_COLUMNS.turbine] == rep.test_wtg].index + truth = rep.true_uplift(mask=treated_activity_mask(test_index, rep.upgrade_timing, window=window)).overall + + logger.info( + "[%s] replicate %d (%s, start %s, %d mo): estimate %+.2f%%, truth %+.2f%%, error %+.2f%% -> %s", + study.mode, + rep.replicate_id, + rep.test_wtg, + pd.Timestamp(rep.treatment_start).date(), + window.months, + 100 * estimate, + 100 * truth, + 100 * (estimate - truth), + rep_dir, + ) + rows.append( + { + "mode": study.mode, + "replicate_id": rep.replicate_id, + "test_wtg": rep.test_wtg, + "treatment_start": rep.treatment_start, + "campaign_months": window.months, + "estimate": estimate, + "truth": truth, + "signed_error": estimate - truth, + "out_dir": str(rep_dir), + } + ) + return rows + + +def inspect_naive_run( + *, + out_root: str | Path | None = None, + data_dir: str | Path | None = None, + n_replicates: int = 3, + delta: float = 0.05, + campaign_months: list[int] | None = None, + toggle_period: pd.Timedelta = DEFAULT_TOGGLE_PERIOD, +) -> pd.DataFrame: + """Run a few naive-ratio prepost and toggle replicates with plots on; return a summary frame. + + :param out_root: output directory; defaults to :func:`default_output_root` / ``inspection_naive`` + :param data_dir: Hill of Towie data/cache dir; defaults to the source package default + :param n_replicates: how many replicate datasets to run per mode (each in its own out dir) + :param delta: injected constant-Cp uplift fraction + :param campaign_months: campaign length(s); the longest is used for each replicate + :param toggle_period: the toggle on/off cycle length (20 on + 20 off = 40 min by default) + :return: per-replicate frame across both modes (mode, test_wtg, estimate, truth, signed_error, ...) + """ + out_dir = (Path(out_root) if out_root is not None else default_output_root()) / "inspection_naive" + out_dir.mkdir(parents=True, exist_ok=True) + campaign_months = campaign_months if campaign_months is not None else [6] + + logger.info("Loading Hill of Towie SCADA %s..%s", DEFAULT_START_DT, DEFAULT_END_DT_EXCL) + scada_df, _metadata_df = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + data_dir=Path(data_dir) if data_dir is not None else None, + ) + + prepost_study = StudyConfig( + mode="prepost", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=campaign_months, + n_replicates=n_replicates, + seed=0, + ) + toggle_study = StudyConfig( + mode="toggle", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=campaign_months, + n_replicates=n_replicates, + toggle_period=toggle_period, + seed=0, + ) + + rows = _inspect_mode(scada_df, study=prepost_study, out_dir=out_dir / "prepost", delta=delta) + rows += _inspect_mode(scada_df, study=toggle_study, out_dir=out_dir / "toggle", delta=delta) + + summary = pd.DataFrame(rows) + summary_path = out_dir / "inspection_summary.csv" + summary.to_csv(summary_path, index=False) + logger.info("Wrote %s\n%s", summary_path, summary.to_string(index=False)) + return summary + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") + inspect_naive_run() diff --git a/benchmarking/baselines/inspect_prepost_feature_ablation.py b/benchmarking/baselines/inspect_prepost_feature_ablation.py new file mode 100644 index 00000000..9f100def --- /dev/null +++ b/benchmarking/baselines/inspect_prepost_feature_ablation.py @@ -0,0 +1,125 @@ +"""Ablate reactive-power / pitch reference features on the F1 hard prepost case. + +A focused follow-up to :mod:`benchmarking.baselines.inspect_prepost_hard_case` (findings F1/F2). +It pins the **same** placebo case (``cp_0pct`` on ``T07`` at 6 months, true uplift 0%) and the +**identical** ``MethodInput``, then runs the R-learner three times while removing reference +features from the SCADA frame before feature-building: + +1. ``full`` — all reference features (the F1 regime); +2. ``no reactive power`` — drop ``wtc_ReactPwr_mean`` from every reference turbine; +3. ``no reactive power, no pitch`` — also drop the three blade-pitch sensors. + +Motivation (the F2 hypothesis): reactive power and pitch are the top propensity features, but the +reactive-power diagnostics show its *control* changes over calendar time, so it (and pitch) may be +acting as proxies for "is this the upgraded season?" — i.e. driving the F1 overlap/positivity +failure rather than carrying physics. Dropping them isolates how much of the prepost bias they own. + +Only the feature set differs between arms; the outcome (test active power), availability, wind speed +and ERA5 features are untouched, and the cross-fitting seed is fixed, so the comparison is +apples-to-apples and deterministic. + +Run it:: + + uv run python -m benchmarking.baselines.inspect_prepost_feature_ablation +""" + +from __future__ import annotations + +import logging +from dataclasses import replace +from typing import TYPE_CHECKING + +import pandas as pd + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, +) +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.inspect_prepost_hard_case import ( + DEFAULT_CAMPAIGN_MONTHS, + DEFAULT_PROFILE, + DEFAULT_TEST_WTG, + _overnight_study, + _pin_case, +) +from benchmarking.baselines.rlearner import RLearnerMethod +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +if TYPE_CHECKING: + from benchmarking.harness import MethodInput + +logger = logging.getLogger(__name__) + +REACTIVE_TAG = "wtc_ReactPwr_mean" +PITCH_TAGS = ("wtc_PitcPosA_mean", "wtc_PitcPosB_mean", "wtc_PitcPosC_mean") + +ARMS: dict[str, tuple[str, ...]] = { + "full (all features)": (), + "no reactive power": (REACTIVE_TAG,), + "no reactive power, no pitch": (REACTIVE_TAG, *PITCH_TAGS), +} + + +def _drop_cols(mi: MethodInput, cols: tuple[str, ...]) -> MethodInput: + """Return a copy of ``mi`` with the given source-native tags removed from every turbine.""" + present = [c for c in cols if c in mi.scada_df.columns] + return replace(mi, scada_df=mi.scada_df.drop(columns=present)) + + +def _make_method(era5_df: pd.DataFrame) -> RLearnerMethod: + """Build an R-learner configured exactly as the inspection driver, plots off (estimate only).""" + return RLearnerMethod( + active_power_col=HOT_COLUMNS.active_power, + wind_speed_col=HOT_COLUMNS.wind_speed, + availability_col=HOT_COLUMNS.availability, + era5_hourly_df=era5_df, + out_dir=None, + save_plots=False, + ) + + +def inspect_prepost_feature_ablation() -> pd.DataFrame: + """Run the three feature-ablation arms on the pinned F1 case; return a tidy comparison frame.""" + study = _overnight_study() + logger.info("Loading Hill of Towie SCADA %s..%s", DEFAULT_START_DT, DEFAULT_END_DT_EXCL) + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + _rep, mi, truth, _window = _pin_case( + scada_df, + study=study, + profile_name=DEFAULT_PROFILE, + test_wtg=DEFAULT_TEST_WTG, + campaign_months=DEFAULT_CAMPAIGN_MONTHS, + ) + + era5_df = build_hot_v0_context(wtg_names=DEFAULT_TURBINE_SUBSET).reanalysis_datasets[0].data + + rows = [] + for label, cols in ARMS.items(): + estimate = _make_method(era5_df).estimate(_drop_cols(mi, cols)).p50_overall + logger.info( + "%-32s estimate %+.3f%% truth %+.3f%% error %+.3f%% (dropped %s)", + label, + 100 * estimate, + 100 * truth, + 100 * (estimate - truth), + list(cols) or "nothing", + ) + rows.append({"arm": label, "estimate": estimate, "truth": truth, "signed_error": estimate - truth}) + + summary = pd.DataFrame(rows) + logger.info("\n%s", summary.to_string(index=False)) + return summary + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO, format="%(message)s") + inspect_prepost_feature_ablation() diff --git a/benchmarking/baselines/inspect_prepost_hard_case.py b/benchmarking/baselines/inspect_prepost_hard_case.py new file mode 100644 index 00000000..1c901043 --- /dev/null +++ b/benchmarking/baselines/inspect_prepost_hard_case.py @@ -0,0 +1,340 @@ +"""Inspect one hard PREPOST case across every method, with plots on. + +A focused investigation driver (findings F1): replay the **exact** overnight prepost study +draws (same ``StudyConfig``/seed/profile), pin a single hard ``(test_wtg, campaign_months)`` +run, and execute naive + power_model (+ v0) on the **identical** ``MethodInput`` with +``save_plots=True``, each into its own subfolder of one timestamped run dir. The default case is +``cp_0pct`` (placebo) on ``T07`` at 6 months — the prepost case where the cross-fit R-learner was +badly biased (~-14%) while naive (~+2%) and v0 (~0%) are fine. The question here is whether the +counterfactual power model, which forms the test-vs-reference contrast the R-learner lacked, also +stays near zero — eyeball the per-method diagnostics side by side to confirm. + +Because the draws are a deterministic function of ``(StudyConfig, seed)`` and the harness builds +one ``MethodInput`` per ``(replicate, window)``, every method here sees the same data the +overnight run scored — only the estimate differs. + +Run it:: + + uv run python -m benchmarking.baselines.inspect_prepost_hard_case + +Outputs land under ``WIND_UP_BENCHMARKING_OUTPUT_DIR``/``inspect_hard_case``/``/`` +(``naive/``, ``power_model/``, ``v0/`` run folders, a ``comparison_summary.csv`` and ``run.log``). +The first run downloads + caches the Hill of Towie SCADA (Zenodo) and ERA5 (Open-Meteo, ``era5`` +group); the ``ml`` group is needed for the power model and v0 needs the wind_up pipeline. +""" + +from __future__ import annotations + +import logging +import time +from pathlib import Path +from typing import TYPE_CHECKING + +import pandas as pd + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TREATMENT_START_RANGE, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, + MIN_PRE_MONTHS, + default_output_root, +) +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.naive_ratio import NaiveRatioMethod +from benchmarking.baselines.overnight_common import start_overnight_run +from benchmarking.baselines.overnight_profiles import overnight_profiles +from benchmarking.baselines.power_model import CURATED_ERA5_EXCLUDE, TUNED_MODEL_PARAMS, PowerModelMethod +from benchmarking.baselines.v0_binned import V0BinnedMethod +from benchmarking.harness import ( + CONDITIONS, + CampaignWindow, + Method, + MethodInput, + MethodOutput, + StudyConfig, + build_replicates, + campaign_windows, + condition_bins, + plot_conditional_uplift, + treated_activity_mask, + window_row_mask, +) +from benchmarking.synthetic import HOT_COLUMNS, HOT_RATED_POWER_KW +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +if TYPE_CHECKING: + from benchmarking.harness.replicates import Replicate + +logger = logging.getLogger(__name__) + +# Match the overnight prepost study exactly so the pinned run is the one its leaderboard scored. +N_REPLICATES = 4 +CAMPAIGN_MONTHS = [3, 6, 12] + +# The default hard case (see module docstring / findings F1). +DEFAULT_PROFILE = "cp_0pct" +DEFAULT_TEST_WTG = "T07" +DEFAULT_CAMPAIGN_MONTHS = 6 + + +def _overnight_study() -> StudyConfig: + """Return the prepost ``StudyConfig`` the overnight run used (so draws are bit-identical).""" + return StudyConfig( + mode="prepost", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=CAMPAIGN_MONTHS, + n_replicates=N_REPLICATES, + seed=0, + ) + + +def _select_replicate(replicates: list[Replicate], test_wtg: str) -> Replicate: + """Return the first replicate drawn for ``test_wtg`` (logging the full draw for transparency).""" + for rep in replicates: + logger.info( + "replicate %d: test_wtg=%s treatment_start=%s", + rep.replicate_id, + rep.test_wtg, + pd.Timestamp(rep.treatment_start).date(), + ) + matches = [rep for rep in replicates if rep.test_wtg == test_wtg] + if not matches: + drawn = sorted({rep.test_wtg for rep in replicates}) + msg = f"no replicate drawn for test_wtg {test_wtg!r}; drawn turbines were {drawn}" + raise ValueError(msg) + return matches[0] + + +def _pin_case( + scada_df: pd.DataFrame, *, study: StudyConfig, profile_name: str, test_wtg: str, campaign_months: int +) -> tuple[Replicate, MethodInput, float, CampaignWindow]: + """Build the pinned replicate, its shared ``MethodInput``, the ground-truth uplift, and the window.""" + profile = overnight_profiles()[profile_name] + replicates = build_replicates(scada_df, profile=profile, study=study) + rep = _select_replicate(replicates, test_wtg) + + windows = campaign_windows( + rep.treatment_start, + min_pre_months=study.min_pre_months, + campaign_months=study.campaign_months, + data_start=scada_df.index.min(), + data_end=scada_df.index.max(), + ) + matching = [w for w in windows if w.months == campaign_months] + if not matching: + available = [w.months for w in windows] + msg = f"campaign_months={campaign_months} not feasible for replicate {rep.replicate_id}; available {available}" + raise ValueError(msg) + window = matching[0] + + syn = rep.synthetic_df + mi = MethodInput( + scada_df=syn.loc[window_row_mask(syn.index, window)], + test_wtg=rep.test_wtg, + upgrade_timing=rep.upgrade_timing, + turbine_col=HOT_COLUMNS.turbine, + ) + test_index = syn.loc[syn[HOT_COLUMNS.turbine] == rep.test_wtg].index + truth = rep.true_uplift(mask=treated_activity_mask(test_index, rep.upgrade_timing, window=window)).overall + logger.info( + "pinned case: profile=%s test_wtg=%s treatment_start=%s campaign_months=%d window=[%s, %s) truth=%+.3f%%", + profile_name, + rep.test_wtg, + pd.Timestamp(rep.treatment_start).date(), + window.months, + window.baseline_start, + window.activity_end, + 100 * truth, + ) + return rep, mi, truth, window + + +def _power_model(out_dir: Path, era5_hourly_df: pd.DataFrame, *, save_plots: bool) -> PowerModelMethod: + """Construct the HoT-configured power model (its default: conditional uplift on).""" + return PowerModelMethod( + columns=HOT_COLUMNS, + baseline_rated_power_kw=HOT_RATED_POWER_KW, + era5_hourly_df=era5_hourly_df, + # Removal-ablation accepted defaults (findings F13). + availability_feature=False, + era5_exclude=CURATED_ERA5_EXCLUDE, + # Issue 12 accepted default (findings F14): looser leaf capacity. + model_params=dict(TUNED_MODEL_PARAMS), + out_dir=out_dir, + save_plots=save_plots, + ) + + +def _build_methods(out_dir: Path, *, include_v0: bool) -> list[Method]: + """Return the methods to inspect, each writing diagnostics (plots on) into its own subfolder. + + One ``power_model`` run folder: with conditional uplift on (the default) it carries the overall + diagnostics plus the conditional CSVs (``conditional/``) and the step-7 implied-shrinkage plot. + """ + context = build_hot_v0_context(wtg_names=DEFAULT_TURBINE_SUBSET) + era5 = context.reanalysis_datasets[0].data + methods: list[Method] = [ + NaiveRatioMethod( + columns=HOT_COLUMNS, + out_dir=out_dir / "naive", + save_plots=True, + ), + _power_model(out_dir / "power_model", era5, save_plots=True), + ] + if include_v0: + methods.append(V0BinnedMethod(context, scratch_dir=out_dir / "v0", save_plots=True)) + return methods + + +def _run_methods( + methods: list[Method], *, mi: MethodInput, truth: float +) -> tuple[pd.DataFrame, dict[str, MethodOutput]]: + """Run every method on the identical input; return the tidy comparison and each method's output. + + The outputs are returned so the power_model conditional estimate (computed here as part of its run) + is reused for the per-condition truth overlay — no second power_model fit. + """ + rows = [] + outputs: dict[str, MethodOutput] = {} + for method in methods: + start = time.perf_counter() + output = method.estimate(mi) + wall_time_s = time.perf_counter() - start + outputs[method.name] = output + estimate = output.p50_overall + logger.info( + "%-12s estimate %+.3f%% truth %+.3f%% error %+.3f%% (%.1fs)", + method.name, + 100 * estimate, + 100 * truth, + 100 * (estimate - truth), + wall_time_s, + ) + rows.append( + { + "method": method.name, + "estimate": estimate, + "truth": truth, + "signed_error": estimate - truth, + "wall_time_s": wall_time_s, + } + ) + return pd.DataFrame(rows), outputs + + +def _plot_conditional_uplift( + rep: Replicate, output: MethodOutput, window: CampaignWindow, *, out_dir: Path, profile_name: str, test_wtg: str +) -> None: + """Overlay the power_model conditional estimate against per-condition truth on the pinned case. + + Reuses the ``MethodOutput`` from the main power_model run (conditional uplift on by default), so the + estimate line is drawn against the known per-condition truth: on a condition-dependent hard case + (or the placebo) it should hug truth. + """ + if output.p50_by_condition is None: + logger.warning("power_model returned no p50_by_condition; skipping conditional uplift plots") + return + + test_index = rep.synthetic_df.loc[rep.synthetic_df[HOT_COLUMNS.turbine] == rep.test_wtg].index + mask = treated_activity_mask(test_index, rep.upgrade_timing, window=window) + truth_by_condition = { + c: rep.true_uplift(mask=mask, by=c, bins=condition_bins(c, rated_power_kw=HOT_RATED_POWER_KW)).by_condition + for c in CONDITIONS + } + # Filter out conditions where true_uplift returned None (should not happen when bins given) + truth_by_condition_clean: dict[str, pd.DataFrame] = { + c: df for c, df in truth_by_condition.items() if df is not None + } + if not truth_by_condition_clean: + logger.warning("No per-condition truth available; skipping conditional uplift plots") + return + + frame = conditional_truth_vs_estimate(output, truth_by_condition_clean, method_name="power_model") + for c in truth_by_condition_clean: + plot_conditional_uplift( + frame, + condition=c, + save_path=out_dir / f"conditional_uplift_{c}.png", + title=f"Conditional uplift ({c}) — profile={profile_name}, wtg={test_wtg} (power_model vs truth)", + ) + logger.info("Wrote conditional uplift plots (power_model vs truth) to %s", out_dir) + + +def conditional_truth_vs_estimate( + output: MethodOutput, truth_by_condition: dict[str, pd.DataFrame], *, method_name: str +) -> pd.DataFrame: + """Shape a method's p50_by_condition + per-condition truth into a plot_conditional_uplift frame.""" + if output.p50_by_condition is None: + msg = "output.p50_by_condition must not be None" + raise ValueError(msg) + frames = [] + bc = output.p50_by_condition + for condition, truth in truth_by_condition.items(): + est = bc[bc["condition"] == condition].set_index("condition_bin")["p50_uplift"] + t = truth.assign(condition_bin=truth["condition_bin"].astype(str)).set_index("condition_bin")["true_uplift"] + bins = est.index.union(t.index) + frames.append( + pd.DataFrame( + { + "method": method_name, + "condition": condition, + "condition_bin": bins, + "mean_estimate": est.reindex(bins).to_numpy(), + "mean_truth": t.reindex(bins).to_numpy(), + } + ) + ) + return pd.concat(frames, ignore_index=True) + + +def inspect_prepost_hard_case( + *, + profile_name: str = DEFAULT_PROFILE, + test_wtg: str = DEFAULT_TEST_WTG, + campaign_months: int = DEFAULT_CAMPAIGN_MONTHS, + out_root: str | Path | None = None, + include_v0: bool = True, +) -> pd.DataFrame: + """Run every method on one pinned hard prepost case with plots on; return the comparison frame. + + :param profile_name: an :func:`overnight_profiles` key (default ``cp_0pct``, the placebo) + :param test_wtg: the test turbine to pin (must be one of the overnight draws) + :param campaign_months: which campaign length of the pinned replicate to inspect + :param out_root: base output dir; defaults to :func:`default_output_root`'s parent + :param include_v0: also run the slow v0 baseline (a full wind_up run for the campaign) + :return: per-method ``estimate``/``truth``/``signed_error``/``wall_time_s`` + """ + study = _overnight_study() + output_root = Path(out_root) if out_root is not None else default_output_root().parent + out_dir = start_overnight_run("inspect_hard_case", study, output_root=output_root) + + logger.info("Loading Hill of Towie SCADA %s..%s", DEFAULT_START_DT, DEFAULT_END_DT_EXCL) + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + + rep, mi, truth, window = _pin_case( + scada_df, study=study, profile_name=profile_name, test_wtg=test_wtg, campaign_months=campaign_months + ) + methods = _build_methods(out_dir, include_v0=include_v0) + summary, outputs = _run_methods(methods, mi=mi, truth=truth) + + _plot_conditional_uplift( + rep, outputs["power_model"], window, out_dir=out_dir, profile_name=profile_name, test_wtg=mi.test_wtg + ) + + summary_path = out_dir / "comparison_summary.csv" + summary.to_csv(summary_path, index=False) + logger.info("Wrote %s\n%s", summary_path, summary.to_string(index=False)) + return summary + + +if __name__ == "__main__": + inspect_prepost_hard_case(include_v0=False) diff --git a/benchmarking/baselines/inspect_short_campaigns.py b/benchmarking/baselines/inspect_short_campaigns.py new file mode 100644 index 00000000..153152b5 --- /dev/null +++ b/benchmarking/baselines/inspect_short_campaigns.py @@ -0,0 +1,168 @@ +"""Short-campaign (1-2 month) exploration: check whether the Issue 9-13 accepted choices hold there. + +The committed benchmark sweeps 3-12 month campaigns; this driver scores 1- and 2-month campaigns +(outside the benchmark grid, so no reference-run merge — oracle + naive anchor the numbers instead +of v0, which is slow) and A/Bs the regime-dependent choices at those lengths. + +Which choices can even flip at short campaigns: + +* **prepost** — a shorter campaign shrinks only the *prediction* window; the training set (the + full pre-changeover baseline) is unchanged, so the fit-side choices (capacity, features) cannot + flip. Only the time-decay weights act on the training side, so prepost trials just those. +* **toggle** — the campaign length scales the training data itself, so capacity + (``min_child_samples``) and the decay weights are genuinely in play. + +Run from the repo root (defaults: both modes, all variants):: + + uv run python -m benchmarking.baselines.inspect_short_campaigns + +Outputs one ``results___.csv`` per run plus a combined +``short_campaign_summary.csv`` / log table of power_model bias/spread/score per +``(mode, variant, campaign_months)`` under ``--output-dir`` +(default ``~/temp/wind-up-benchmarking/short_campaigns``). +""" + +from __future__ import annotations + +import argparse +import logging +from pathlib import Path +from typing import Any, Literal, cast + +import pandas as pd + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TREATMENT_START_RANGE, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, + MIN_PRE_MONTHS, +) +from benchmarking.baselines.example_toggle_study import DEFAULT_TOGGLE_PERIOD +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.naive_ratio import NaiveRatioMethod +from benchmarking.baselines.overnight_profiles import overnight_profiles +from benchmarking.baselines.study_power_model_compare import _make_power_model +from benchmarking.harness import Method, StudyConfig, leaderboard, score_study +from benchmarking.harness.example_hot_study import OracleMethod +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +CAMPAIGN_MONTHS = [1, 2] +N_REPLICATES = 4 +SEED = 0 +PROFILES = ("cp_0pct", "cp_plus_3pct") # placebo (bias/spread) + a plain recovery check +_DEFAULT_OUTPUT_DIR = Path.home() / "temp" / "wind-up-benchmarking" / "short_campaigns" + +# Variant name -> (modes it is meaningful for, PowerModelMethod overrides). The "default" anchor +# also scores oracle + naive for context. See the module docstring for why prepost trials only +# the decay weights. +VARIANTS: dict[str, tuple[tuple[str, ...], dict[str, Any]]] = { + "default": (("prepost", "toggle"), {}), + "hl90": (("prepost", "toggle"), {"adaptive_time_decay": False, "time_decay_half_life_days": 90}), + "hl365": (("prepost", "toggle"), {"adaptive_time_decay": False, "time_decay_half_life_days": 365}), + "mcs200": (("toggle",), {"model_params": {"min_child_samples": 200}}), +} + + +def _study(mode: str) -> StudyConfig: + return StudyConfig( + mode=cast("Literal['prepost', 'toggle']", mode), + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=CAMPAIGN_MONTHS, + toggle_period=DEFAULT_TOGGLE_PERIOD if mode == "toggle" else None, + n_replicates=N_REPLICATES, + seed=SEED, + ) + + +def run_variant( + mode: str, + variant: str, + out_dir: Path, + *, + scada_df: pd.DataFrame, + era5_hourly_df: pd.DataFrame, + profiles: dict[str, list], +) -> pd.DataFrame: + """Score one (mode, variant) over the short-campaign grid; anchor methods only on ``default``.""" + overrides = VARIANTS[variant][1] + frames = [] + for profile_name, profile in profiles.items(): + methods: list[Method] = [] + if variant == "default": + methods.append(OracleMethod(scada_df)) + methods.append( + NaiveRatioMethod( + columns=HOT_COLUMNS, + out_dir=out_dir / "naive_runs", + ) + ) + methods.append( + _make_power_model(out_dir / variant / profile_name, era5_hourly_df=era5_hourly_df, overrides=overrides) + ) + logger.info("Scoring %s / %s / %s", mode, variant, profile_name) + results = score_study(scada_df, profile=profile, methods=methods, study=_study(mode), profile_name=profile_name) + results.to_csv(out_dir / f"results_{mode}_{variant}_{profile_name}.csv", index=False) + frames.append(results.assign(variant=variant, mode=mode)) + return pd.concat(frames, ignore_index=True) + + +def summarise(all_results: pd.DataFrame, out_dir: Path) -> None: + """Write/log power_model bias/spread/score per (mode, variant, profile, campaign) + the anchors.""" + rows = [] + for (mode, variant), chunk in all_results.groupby(["mode", "variant"]): + lb = leaderboard(chunk) + lb = lb.assign(mode=mode, variant=variant) + rows.append(lb) + summary = pd.concat(rows, ignore_index=True) + cols = ["mode", "variant", "method", "profile", "campaign_months", "bias", "spread", "score"] + summary = summary[cols].sort_values(["mode", "profile", "campaign_months", "method", "variant"]) + summary.to_csv(out_dir / "short_campaign_summary.csv", index=False) + show = summary.copy() + for col in ("bias", "spread", "score"): + show[col] = (100 * show[col]).round(3) + logger.info("Short-campaign summary [pp]:\n%s", show.to_string(index=False)) + + +def main() -> None: + """Run the short-campaign exploration for the requested modes/variants.""" + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--modes", nargs="+", choices=["prepost", "toggle"], default=["prepost", "toggle"]) + parser.add_argument("--variants", nargs="+", choices=sorted(VARIANTS), default=None) + parser.add_argument("--output-dir", type=Path, default=_DEFAULT_OUTPUT_DIR) + args = parser.parse_args() + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s", force=True) + + out_dir = args.output_dir.expanduser() + out_dir.mkdir(parents=True, exist_ok=True) + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + context = build_hot_v0_context(wtg_names=DEFAULT_TURBINE_SUBSET) + era5 = context.reanalysis_datasets[0].data + profiles = {name: overnight_profiles()[name] for name in PROFILES} + + all_results = [] + for mode in args.modes: + for variant, (variant_modes, _) in VARIANTS.items(): + if args.variants is not None and variant not in args.variants: + continue + if mode not in variant_modes: + continue + all_results.append( + run_variant(mode, variant, out_dir, scada_df=scada_df, era5_hourly_df=era5, profiles=profiles) + ) + summarise(pd.concat(all_results, ignore_index=True), out_dir) + + +if __name__ == "__main__": + main() diff --git a/benchmarking/baselines/inspect_v0_run.py b/benchmarking/baselines/inspect_v0_run.py new file mode 100644 index 00000000..2cfd3ab7 --- /dev/null +++ b/benchmarking/baselines/inspect_v0_run.py @@ -0,0 +1,158 @@ +"""Manual inspection driver: run a few v0 replicates with wind_up plots saved to file. + +A runnable companion to the ``slow`` end-to-end test +(``tests/benchmarking/baselines/test_v0_end_to_end.py``): same real-data setup, but instead of +asserting it runs a handful of replicates with ``save_plots=True``, each in its **own output +directory**, so the full set of wind_up diagnostic plots (power curves, detrend, data coverage, +pre/post comparisons, ...) can be eyeballed to confirm the baseline is wired correctly. It also +writes an ``inspection_summary.csv`` of recovered P50 vs injected truth per replicate. + +Each replicate is built and scored exactly as :func:`benchmarking.harness.score_study` would +(same windowing and ground truth), so what you inspect is the real scored path. + +Run it:: + + uv run python -m benchmarking.baselines.inspect_v0_run + +First run downloads and caches the Hill of Towie v2 SCADA (Zenodo) and ERA5 (Open-Meteo; needs +the ``era5`` optional dependency group). Each replicate's plots land under +``/inspection/replicate__/v0__/plots``. +""" + +from __future__ import annotations + +import logging +from pathlib import Path + +import pandas as pd + +from benchmarking.baselines.example_prepost_study import default_output_root +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.v0_binned import V0BinnedMethod +from benchmarking.harness import ( + MethodInput, + StudyConfig, + build_replicates, + campaign_windows, + treated_activity_mask, + window_row_mask, +) +from benchmarking.synthetic import HOT_COLUMNS, ConstantCpChange +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# The full 2016-2020 stable window, same as the driver: treatment drawn per replicate from +# 2018-01-01..2020-01-01, with a 24-month baseline (so the earliest upgrade's baseline starts at +# the 2016-01-01 data start and v0's detrend is fully covered). +DEFAULT_START_DT = pd.Timestamp("2016-01-01", tz="UTC") +DEFAULT_END_DT_EXCL = pd.Timestamp("2021-01-01", tz="UTC") +DEFAULT_WTG_NUMBERS = [1, 3, 4, 7] +DEFAULT_TURBINE_SUBSET = [f"T{x:02d}" for x in DEFAULT_WTG_NUMBERS] +DEFAULT_TREATMENT_START_RANGE = (pd.Timestamp("2018-01-01", tz="UTC"), pd.Timestamp("2019-12-31 23:50", tz="UTC")) +MIN_PRE_MONTHS = 24 + + +def inspect_v0_run( + *, + out_root: str | Path | None = None, + data_dir: str | Path | None = None, + n_replicates: int = 3, + delta: float = 0.05, + campaign_months: list[int] | None = None, +) -> pd.DataFrame: + """Run ``n_replicates`` v0 estimates with plots saved per replicate; return a summary frame. + + :param out_root: output directory; defaults to :func:`default_output_root` / ``inspection`` + :param data_dir: Hill of Towie data/cache dir; defaults to the source package default + :param n_replicates: how many replicate datasets to run (each in its own out dir) + :param delta: injected constant-Cp uplift fraction + :param campaign_months: campaign length(s); the longest is used for each replicate's plots + :return: per-replicate frame of test_wtg, treatment_start, campaign_months, estimate, truth, signed_error + """ + out_dir = (Path(out_root) if out_root is not None else default_output_root()) / "inspection" + out_dir.mkdir(parents=True, exist_ok=True) + campaign_months = campaign_months if campaign_months is not None else [6] + + logger.info("Loading Hill of Towie SCADA %s..%s", DEFAULT_START_DT, DEFAULT_END_DT_EXCL) + context = build_hot_v0_context(wtg_names=DEFAULT_TURBINE_SUBSET) + scada_df, _metadata_df = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + data_dir=Path(data_dir) if data_dir is not None else None, + ) + study = StudyConfig( + mode="prepost", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=campaign_months, + n_replicates=n_replicates, + seed=0, + ) + replicates = build_replicates(scada_df, profile=[ConstantCpChange(delta=delta)], study=study) + + rows = [] + for rep in replicates: + windows = campaign_windows( + rep.treatment_start, + min_pre_months=study.min_pre_months, + campaign_months=study.campaign_months, + data_start=scada_df.index.min(), + data_end=scada_df.index.max(), + ) + if not windows: + logger.warning("replicate %d (%s): no feasible campaign window, skipping", rep.replicate_id, rep.test_wtg) + continue + window = windows[-1] # the longest campaign -> richest plots + + syn = rep.synthetic_df + mi = MethodInput( + scada_df=syn.loc[window_row_mask(syn.index, window)], + test_wtg=rep.test_wtg, + upgrade_timing=rep.upgrade_timing, + turbine_col=HOT_COLUMNS.turbine, + ) + rep_dir = out_dir / f"replicate_{rep.replicate_id:02d}_{rep.test_wtg}" + method = V0BinnedMethod(context, scratch_dir=rep_dir, save_plots=True) + estimate = method.estimate(mi).p50_overall + + test_index = syn.loc[syn[HOT_COLUMNS.turbine] == rep.test_wtg].index + truth = rep.true_uplift(mask=treated_activity_mask(test_index, rep.upgrade_timing, window=window)).overall + + logger.info( + "replicate %d (%s, start %s, %d mo): estimate %+.2f%%, truth %+.2f%%, error %+.2f%% -> %s", + rep.replicate_id, + rep.test_wtg, + pd.Timestamp(rep.treatment_start).date(), + window.months, + 100 * estimate, + 100 * truth, + 100 * (estimate - truth), + rep_dir, + ) + rows.append( + { + "replicate_id": rep.replicate_id, + "test_wtg": rep.test_wtg, + "treatment_start": rep.treatment_start, + "campaign_months": window.months, + "estimate": estimate, + "truth": truth, + "signed_error": estimate - truth, + "out_dir": str(rep_dir), + } + ) + + summary = pd.DataFrame(rows) + summary_path = out_dir / "inspection_summary.csv" + summary.to_csv(summary_path, index=False) + logger.info("Wrote %s\n%s", summary_path, summary.to_string(index=False)) + return summary + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") + inspect_v0_run() diff --git a/benchmarking/baselines/migrate_toggle_baseline_v2_to_v3.py b/benchmarking/baselines/migrate_toggle_baseline_v2_to_v3.py new file mode 100644 index 00000000..c44d0f73 --- /dev/null +++ b/benchmarking/baselines/migrate_toggle_baseline_v2_to_v3.py @@ -0,0 +1,101 @@ +"""One-off: split the single v2 toggle benchmark into a portable + a per-platform v3 pair (F30). + +The v2 file was recorded on a Windows laptop and holds both methods. ``power_model``'s cells are +machine-specific, so they can only ever be re-derived on that machine — hence a split rather than a +re-record. ``toggle_specialist``'s cells are portable (verified at ~5e-07 pp on Linux), so they +become the shared baseline unchanged. + +Delete this script once both platform baselines are recorded and committed. + + uv run python -m benchmarking.baselines.migrate_toggle_baseline_v2_to_v3 +""" + +from __future__ import annotations + +import argparse +import json +import logging +from pathlib import Path +from typing import Any + +import pandas as pd + +from benchmarking.baselines.study_toggle_methods_compare import ( + _BASELINE_DIR, + _BASELINE_SCHEMA, + _portable_methods, + baseline_paths, +) + +logger = logging.getLogger(__name__) + +_V2_SCHEMA = "toggle_methods_compare_baseline_v2" +_V2_NAME = "study_toggle_methods_compare_baseline.json" +# The v2 file predates the machine fingerprint. Its platform is known (it was recorded on the Windows +# laptop); the rest cannot be recovered and are written as null, which _warn_on_fingerprint_mismatch +# reads as "no claim" rather than guessing. +_RECORDED_PLATFORM = "win32" +_UNRECOVERABLE: dict[str, Any] = {"cpu_count": None, "python_version": None, "lightgbm_version": None} + + +def _split(doc: dict[str, Any]) -> tuple[pd.DataFrame, pd.DataFrame]: + """Split v2 cells into (portable, machine-specific) by method.""" + cells = pd.DataFrame(doc["cells"]) + portable_names = _portable_methods() + return cells[cells["method"].isin(portable_names)], cells[~cells["method"].isin(portable_names)] + + +def _write(path: Path, *, cells: pd.DataFrame, doc: dict[str, Any], platform: str) -> None: + """Write one v3 file, carrying the v2 provenance forward rather than restamping it.""" + out = { + "schema": _BASELINE_SCHEMA, + "recorded_utc": doc["recorded_utc"], + "git_commit": doc["git_commit"], + "platform": platform, + **_UNRECOVERABLE, + "n_replicates": doc["n_replicates"], + "seed": doc["seed"], + "campaign_weeks": doc["campaign_weeks"], + "profiles": doc["profiles"], + "methods": sorted(cells["method"].unique()), + "cells": cells.to_dict(orient="records"), + } + path.write_text(json.dumps(out, indent=2) + "\n") + logger.info("Wrote %s (%d cells, methods %s)", path.name, len(cells), out["methods"]) + + +def migrate(baseline_dir: Path | None = None) -> None: + """Split the committed v2 file into the v3 portable + win32 pair.""" + directory = _BASELINE_DIR if baseline_dir is None else baseline_dir + v2_path = directory / _V2_NAME + doc = json.loads(v2_path.read_text()) + if doc.get("schema") != _V2_SCHEMA: + msg = f"{v2_path} has schema {doc.get('schema')!r}, expected {_V2_SCHEMA!r}; nothing to migrate" + raise ValueError(msg) + + portable, machine_specific = _split(doc) + if portable.empty or machine_specific.empty: + msg = ( + f"expected both portable and machine-specific cells in {v2_path}; " + f"got {len(portable)} and {len(machine_specific)}" + ) + raise ValueError(msg) + + portable_path, _ = baseline_paths(directory) + _, win32_path = baseline_paths(directory, platform=_RECORDED_PLATFORM) + _write(portable_path, cells=portable, doc=doc, platform=_RECORDED_PLATFORM) + _write(win32_path, cells=machine_specific, doc=doc, platform=_RECORDED_PLATFORM) + logger.info("Migration done. `git rm %s`, then record this machine's own platform baseline.", _V2_NAME) + + +def main() -> None: + """Run the migration.""" + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--baseline-dir", type=Path, default=None, help="where the baseline JSONs live") + args = parser.parse_args() + logging.basicConfig(level=logging.INFO, format="%(levelname)s %(message)s", force=True) + migrate(args.baseline_dir.expanduser() if args.baseline_dir else None) + + +if __name__ == "__main__": + main() diff --git a/benchmarking/baselines/naive_ratio.py b/benchmarking/baselines/naive_ratio.py new file mode 100644 index 00000000..6ef929b9 --- /dev/null +++ b/benchmarking/baselines/naive_ratio.py @@ -0,0 +1,506 @@ +"""A deliberately naive energy-ratio uplift method behind the harness ``Method`` seam. + +``NaiveRatioMethod`` is an honest, independent baseline that shares no code with v0 and has +no wind_up dependency. For a set of rows it forms the test-to-reference ratio + + rho(rows) = sum(test power) / sum(reference total power) + +over *used* timestamps and estimates uplift as the ratio-of-ratios +``rho(treated) / rho(baseline) - 1``. Its only error source on the synthetic data is genuine +pre/post covariate shift -- it applies no conditioning by design -- so it is the "what if you +don't condition at all" leaderboard floor. + +**Used timestamps** require every turbine (test and references) to be available (an availability +counter at a full period) and have finite power — a down turbine on either side of the ratio would +otherwise bias it. The availability column is therefore **required** (the lack of downtime +filtering was a real oversight). Only the active-power column enters the ``rho`` *computation*; the +availability column is used solely for row selection (cause, not effect), so the estimate still +never conditions on the test turbine's post-treatment wind speed (design-note §3). It speaks the +data source's own column names and has no wind_up dependency. + +Each run writes a per-run folder ``naive___/`` (v0-style naming) +under ``out_dir`` (a temp dir by default), holding a per-segment data-stats CSV, a headline +results CSV, and -- when ``save_plots`` -- three diagnostic plots (a test-vs-reference scatter, +a per-segment daily-ratio timeseries, and a per-segment used-data-coverage timeseries). The rich stats let a human +confirm the right data was received and interpreted: the headline uplift is re-derivable from +the stats CSV as ``rho = used_test_mwh / used_ref_total_mwh`` per segment. +""" + +from __future__ import annotations + +import tempfile +from dataclasses import dataclass, replace +from pathlib import Path +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +from matplotlib.ticker import PercentFormatter + +from benchmarking.baselines.filtering import NormalOperationFilter +from benchmarking.diagnostics import DiagnosticContext, stages, write_common_diagnostics, write_run_config +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.harness.toggle import ToggleRowSets, is_toggle, resolve_toggle, toggle_upgrade_start +from benchmarking.synthetic import ToggleSchedule + +if TYPE_CHECKING: + import numpy.typing as npt + + from benchmarking.synthetic import ColumnSchema + +_SEGMENTS = ("all", "baseline", "upgraded") +_MIN_POINTS_FOR_TIMEBASE = 2 + + +def _infer_timebase(index: pd.DatetimeIndex) -> pd.Timedelta: + """Infer the analysis timebase as the median spacing of the sorted unique timestamps.""" + unique = pd.DatetimeIndex(pd.unique(index)).sort_values() + if len(unique) < _MIN_POINTS_FOR_TIMEBASE: + return pd.Timedelta(minutes=10) + return pd.Timedelta(np.median(np.diff(unique.to_numpy()))) + + +def _wide_column(scada_df: pd.DataFrame, *, turbine_col: str, value_col: str) -> pd.DataFrame: + """Pivot long SCADA to a timestamp x turbine table of ``value_col`` (NaN where missing).""" + tmp = scada_df[[turbine_col, value_col]].copy() + tmp["_ts"] = scada_df.index + return tmp.pivot_table( + index="_ts", + columns=turbine_col, + values=value_col, + aggfunc="first", + ) + + +def restrict_to_campaign(mi: MethodInput, *, toggle_campaign_only: bool) -> MethodInput: + """Drop pre-campaign rows for a toggle campaign so the on/off comparison shares a distribution. + + For toggle, the harness window also carries the pre-campaign baseline, whose distribution + differs from the campaign and reintroduces the covariate shift toggling exists to avoid. When + ``toggle_campaign_only`` (the default), restrict a toggle input to records at/after the toggle + start, leaving only the interleaved on/off blocks. No-op for prepost (its baseline *is* the + pre-campaign data) and when the flag is off. + """ + timing = mi.upgrade_timing + if not (toggle_campaign_only and isinstance(timing, ToggleSchedule) and timing.start is not None): + return mi + return replace(mi, scada_df=mi.scada_df.loc[mi.scada_df.index >= timing.start]) + + +@dataclass +class NaiveRatioMethod: + """Pluggable naive energy-ratio baseline (prepost and toggle). + + :param columns: **required** source-native column schema — the single source of the method's + column names. It reads the ``active_power`` role (the only signal in the ``rho`` computation; + the turbine identifier comes from the seam, ``MethodInput``) and the ``availability`` role + (the required downtime filter, applied to the test turbine and every reference; rows below a + full period of availability are dropped, so downtime filtering can never be silently skipped). + The remaining roles feed only the diagnostics, never the ``rho`` computation. + :param name: method name shown in the leaderboard + :param out_dir: where per-run folders are written; a temp dir when ``None`` + :param save_plots: also write the diagnostic plots under ``/plots`` + :param timebase: analysis timebase; inferred from the data when ``None`` + :param toggle_campaign_only: for a toggle campaign, fit only on the interleaved on/off blocks + (drop the pre-campaign baseline) so on and off share a wind distribution; no-op for prepost + """ + + columns: ColumnSchema + name: str = "naive_ratio" + out_dir: Path | None = None + save_plots: bool = False + timebase: pd.Timedelta | None = None + toggle_campaign_only: bool = True + + def __post_init__(self) -> None: + """Validate ``columns`` names every role this method reads.""" + self.columns.require_roles(("active_power", "availability")) + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Estimate the test turbine's P50 uplift for one campaign and write diagnostics.""" + mi = restrict_to_campaign(mi, toggle_campaign_only=self.toggle_campaign_only) + wide = _wide_column(mi.scada_df, turbine_col=mi.turbine_col, value_col=self.columns.active_power) + test = mi.test_wtg + refs = [c for c in wide.columns if c != test] + if not refs: + msg = ( + f"no reference turbines available for test_wtg {test!r}: scada_df contains only " + f"{list(wide.columns)}. The naive ratio method needs at least one reference turbine." + ) + raise ValueError(msg) + + if self.columns.availability not in mi.scada_df.columns: + msg = ( + f"the availability column {self.columns.availability!r} (columns.availability) is not in " + f"scada_df; the downtime filter is required for the naive method and cannot be skipped." + ) + raise ValueError(msg) + + timebase = self.timebase if self.timebase is not None else _infer_timebase(mi.scada_df.index) + rows = resolve_toggle(mi.upgrade_timing, wide.index) + baseline = rows.campaign_baseline if self.toggle_campaign_only else rows.training_baseline + test_pw = wide[test].to_numpy(dtype=float) + ref_total = wide[refs].sum(axis=1).to_numpy(dtype=float) + used = self._used_mask(mi, wide=wide, test=test, refs=refs, timebase=timebase).to_numpy() + + rho_base = _rho(test_pw, ref_total, used & baseline) + rho_up = _rho(test_pw, ref_total, used & rows.upgraded) + recoverable = np.isfinite(rho_base) and rho_base != 0 and np.isfinite(rho_up) + uplift = rho_up / rho_base - 1.0 if recoverable else np.nan + + stats = _segment_stats( + mi, + wide=wide, + used=used, + toggle_rows=rows, + toggle_campaign_only=self.toggle_campaign_only, + refs=refs, + timebase=timebase, + active_power_col=self.columns.active_power, + ) + self._write_outputs( + mi, + wide=wide, + stats=stats, + used=used, + rho_base=rho_base, + rho_up=rho_up, + uplift=uplift, + n_refs=len(refs), + timebase=timebase, + ) + return MethodOutput(p50_overall=float(uplift)) + + def _used_mask( + self, mi: MethodInput, *, wide: pd.DataFrame, test: str, refs: list[str], timebase: pd.Timedelta + ) -> pd.Series: + """Complete-case timestamps that also pass downtime filtering on the test turbine and every reference. + + Returns a bool Series on ``wide.index``. Every turbine (test and references) must be + available (counter >= a full period) and have finite power — a down turbine on either side + of the ratio is therefore excluded. The test turbine additionally goes through the shared + :class:`NormalOperationFilter` (the same downtime + finite-power logic the R-learner uses; + the stuck filter is left off here as the ratio sums raw power rather than fitting a model). + """ + turbines = [test, *refs] + complete = wide[turbines].notna().all(axis=1) + + full = timebase.total_seconds() + avail = _wide_column(mi.scada_df, turbine_col=mi.turbine_col, value_col=self.columns.availability).reindex( + index=wide.index, columns=turbines + ) + all_available = (avail >= full).all(axis=1) + + test_rows = mi.scada_df[mi.scada_df[mi.turbine_col] == test] + test_keep = ( + NormalOperationFilter( + active_power_col=self.columns.active_power, + availability_col=self.columns.availability, + apply_stuck_filter=False, + ) + .keep_mask(test_rows, timebase=timebase) + .reindex(wide.index, fill_value=False) + ) + return complete & all_available & test_keep + + def _write_outputs( + self, + mi: MethodInput, + *, + wide: pd.DataFrame, + stats: pd.DataFrame, + used: np.ndarray, + rho_base: float, + rho_up: float, + uplift: float, + n_refs: int, + timebase: pd.Timedelta, + ) -> None: + """Write the data-stats CSV, the headline results CSV and (optionally) the plots.""" + upgrade_start = toggle_upgrade_start(mi.upgrade_timing, wide.index) + last_dt = wide.index.max() + run_name = f"naive_{mi.test_wtg}_{upgrade_start:%Y%m%d}_{last_dt:%Y%m%d}" + out_root = Path(self.out_dir) if self.out_dir is not None else Path(tempfile.mkdtemp(prefix="naive_")) + run_dir = out_root / run_name + run_dir.mkdir(parents=True, exist_ok=True) + ts = pd.Timestamp.utcnow().strftime("%Y%m%d_%H%M%S_%f") + + stats.to_csv(run_dir / f"{run_name}_data_stats_{ts}.csv", index=False) + + mode = "toggle" if is_toggle(mi.upgrade_timing) else "prepost" + used_base = int(stats.loc[stats["segment"] == "baseline", "n_used_timestamps"].iloc[0]) + used_up = int(stats.loc[stats["segment"] == "upgraded", "n_used_timestamps"].iloc[0]) + results = pd.DataFrame( + [ + { + "test_wtg": mi.test_wtg, + "mode": mode, + "n_turbines": wide.shape[1], + "n_refs": n_refs, + "ratio_baseline": rho_base, + "ratio_upgraded": rho_up, + "uplift_frc": uplift, + "n_used_timestamps_baseline": used_base, + "n_used_timestamps_upgraded": used_up, + "time_calculated": pd.Timestamp.utcnow(), + } + ] + ) + results.to_csv(run_dir / f"{run_name}_results_{ts}.csv", index=False) + + if self.save_plots: + _save_plots( + run_dir / "plots", + wide=wide, + mi=mi, + test=mi.test_wtg, + used=used, + timebase=timebase, + active_power_col=self.columns.active_power, + toggle_campaign_only=self.toggle_campaign_only, + ) + self._write_shared_diagnostics(mi, run_dir=run_dir, wide=wide, timebase=timebase) + + def _write_shared_diagnostics( + self, mi: MethodInput, *, run_dir: Path, wide: pd.DataFrame, timebase: pd.Timedelta + ) -> None: + """Emit the shared cross-method diagnostics (coverage/curves/histograms) and the run config.""" + # ``wide`` (a pivot) drops all-NaN timestamps, so align the masks to the full unique index + # the DiagnosticContext uses (timestamps absent from ``wide`` are simply not used). + index = pd.DatetimeIndex(pd.unique(mi.scada_df.index)).sort_values() + test, refs = mi.test_wtg, [c for c in wide.columns if c != mi.test_wtg] + used_series = self._used_mask(mi, wide=wide, test=test, refs=refs, timebase=timebase) + used = used_series.reindex(index, fill_value=False).to_numpy() + treated = resolve_toggle(mi.upgrade_timing, index).upgraded.astype(bool) + ctx = DiagnosticContext( + run_dir=run_dir, + test_wtg=mi.test_wtg, + turbine_col=mi.turbine_col, + columns=self.columns, + scada_df=mi.scada_df, + treated_ts=treated, + used_ts=used, + timebase=timebase, + mode="toggle" if is_toggle(mi.upgrade_timing) else "prepost", + era5_df=None, + ) + write_common_diagnostics(ctx) + params = { + "active_power_col": self.columns.active_power, + "availability_col": self.columns.availability, + "toggle_campaign_only": self.toggle_campaign_only, + } + write_run_config(ctx, method_name=self.name, method_params=params) + + +def _rho(test_pw: npt.NDArray[np.float64], ref_total: npt.NDArray[np.float64], mask: npt.NDArray[np.bool_]) -> float: + """Test-to-reference ratio over ``mask``: sum(test) / sum(ref_total). NaN if degenerate.""" + if not mask.any(): + return float("nan") + denom = ref_total[mask].sum() + if denom == 0: + return float("nan") + return float(test_pw[mask].sum() / denom) + + +def _segment_stats( + mi: MethodInput, + *, + wide: pd.DataFrame, + used: npt.NDArray[np.bool_], + toggle_rows: ToggleRowSets, + toggle_campaign_only: bool, + refs: list[str], + timebase: pd.Timedelta, + active_power_col: str, +) -> pd.DataFrame: + """Build the per-segment (all/baseline/upgraded) diagnostics table.""" + test = mi.test_wtg + test_pw = wide[test].to_numpy(dtype=float) + ref_total = wide[refs].sum(axis=1).to_numpy(dtype=float) + n_turbines = wide.shape[1] + timebase_hours = timebase / pd.Timedelta(hours=1) + + row_rows = resolve_toggle(mi.upgrade_timing, mi.scada_df.index) + row_power = mi.scada_df[active_power_col].to_numpy(dtype=float) + + ts_baseline = toggle_rows.campaign_baseline if toggle_campaign_only else toggle_rows.training_baseline + row_baseline = row_rows.campaign_baseline if toggle_campaign_only else row_rows.training_baseline + ts_masks = {"all": np.ones(len(wide), dtype=bool), "baseline": ts_baseline, "upgraded": toggle_rows.upgraded} + row_masks = {"all": np.ones(len(mi.scada_df), dtype=bool), "baseline": row_baseline, "upgraded": row_rows.upgraded} + + rows = [] + for segment in _SEGMENTS: + ts_mask = ts_masks[segment] + row_mask = row_masks[segment] + seg_ts = wide.index[ts_mask] + seg_used = used & ts_mask + n_used = int(seg_used.sum()) + + if len(seg_ts): + first, last = seg_ts.min(), seg_ts.max() + expected_ts = round((last - first) / timebase) + 1 + else: + first = last = pd.NaT + expected_ts = 0 + expected_rows = n_turbines * expected_ts + + n_rows = int(row_mask.sum()) + n_power_finite = int(np.isfinite(row_power[row_mask]).sum()) + + used_test = test_pw[seg_used] + used_ref = ref_total[seg_used] + rows.append( + { + "segment": segment, + "first_timestamp": first, + "last_timestamp": last, + "n_turbines": n_turbines, + "expected_timestamps": expected_ts, + "n_rows": n_rows, + "expected_rows": expected_rows, + "rows_data_coverage": n_rows / expected_rows if expected_rows else np.nan, + "n_power_finite_rows": n_power_finite, + "power_finite_coverage": n_power_finite / expected_rows if expected_rows else np.nan, + "n_used_timestamps": n_used, + "used_data_coverage": n_used / expected_ts if expected_ts else np.nan, + "used_test_mean_power_kw": float(used_test.mean()) if n_used else np.nan, + "used_test_mwh": float(used_test.sum()) * timebase_hours / 1000.0 if n_used else np.nan, + "used_ref_total_mean_power_kw": float(used_ref.mean()) if n_used else np.nan, + "used_ref_total_mwh": float(used_ref.sum()) * timebase_hours / 1000.0 if n_used else np.nan, + } + ) + return pd.DataFrame(rows) + + +def _daily_segment_ratio( + index: pd.DatetimeIndex, + test_pw: npt.NDArray[np.float64], + ref_total: npt.NDArray[np.float64], + seg_mask: npt.NDArray[np.bool_], +) -> pd.Series: + """Daily sum-based test/reference ratio (Sum test / Sum ref) over ``seg_mask`` rows; NaN on empty days. + + This matches the method's own ``rho`` definition (a ratio of sums, not a mean of per-timestamp + ratios), so the daily series fluctuates around the scalar ``rho`` the estimate uses instead of + blowing up on low-wind timestamps. + """ + test = pd.Series(np.where(seg_mask, test_pw, np.nan), index=index) + ref = pd.Series(np.where(seg_mask, ref_total, np.nan), index=index) + return test.resample("1D").sum(min_count=1) / ref.resample("1D").sum(min_count=1) + + +def _expected_per_day(index: pd.DatetimeIndex, timebase: pd.Timedelta) -> pd.Series: + """Daily count of timestamps the analysis timebase grid expects between the data's first and last.""" + grid = pd.date_range(index.min(), index.max(), freq=timebase) + return pd.Series(1.0, index=grid).resample("1D").sum() + + +def _daily_segment_coverage( + index: pd.DatetimeIndex, + used: npt.NDArray[np.bool_], + seg_mask: npt.NDArray[np.bool_], + expected_per_day: pd.Series, +) -> pd.Series: + """Daily used-data coverage in [0, 1], as a fraction of the day's expected timestamps. + + Numerator: complete-case timestamps (test and every reference finite) assigned to this segment + each day. Denominator: the day's expected timestamp count on the analysis timebase grid, which + is shared across segments. So the two segments' coverages sum to the day's overall complete-case + coverage, and under toggle each segment is capped near the duty cycle (~50%) of slots it can ever + occupy. NaN on days the grid does not reach. + """ + used_seg = pd.Series((used & seg_mask).astype(float), index=index) + daily_used = used_seg.resample("1D").sum() + return daily_used / expected_per_day.reindex(daily_used.index) + + +def _save_plots( + plots_dir: Path, + *, + wide: pd.DataFrame, + mi: MethodInput, + test: str, + used: np.ndarray, + timebase: pd.Timedelta, + active_power_col: str, + toggle_campaign_only: bool, +) -> None: + """Write the scatter, ratio-timeseries and used-coverage-timeseries diagnostic plots (by stage). + + ``used`` is the method's real downtime-filtered mask (test + every reference passing the + availability/finite filter), so the scatter shows only the rows the estimate actually uses. + ``toggle_campaign_only`` picks the same baseline the estimate used (strict off-blocks when set, + the lenient pre-campaign U off baseline when not), so the plots never disagree with the headline. + """ + refs = [c for c in wide.columns if c != test] + toggle_rows = resolve_toggle(mi.upgrade_timing, wide.index) + baseline_mask = toggle_rows.campaign_baseline if toggle_campaign_only else toggle_rows.training_baseline + test_pw = wide[test].to_numpy(dtype=float) + ref_total = wide[refs].sum(axis=1).to_numpy(dtype=float) + upgrade_start = toggle_upgrade_start(mi.upgrade_timing, wide.index) + segments = ( + ("baseline", used & baseline_mask, "C0"), + ("upgraded", used & toggle_rows.upgraded, "C1"), + ) + + # 1) scatter of test vs reference-total power, baseline/upgraded coloured, with rho slopes. + fig, ax = plt.subplots(figsize=(7, 7)) + for label, seg, color in segments: + ax.scatter(ref_total[seg], test_pw[seg], s=8, alpha=0.4, color=color, label=label) + rho = _rho(test_pw, ref_total, seg) + if np.isfinite(rho) and seg.any(): + x_max = float(np.nanmax(ref_total[seg])) + ax.plot([0, x_max], [0, rho * x_max], color=color, linewidth=1.5) + ax.set_xlabel(f"sum of reference {active_power_col} [kW]") + ax.set_ylabel(f"{active_power_col} @ {test} [kW]") + ax.set_title(f"{test}: test vs reference-total power") + ax.grid(visible=True, alpha=0.3) + ax.legend() + fig.tight_layout() + _save(fig, plots_dir / stages.UPLIFT_INPUTS / f"{test}_scatter.png") + + # 2) daily sum-based test/ref ratio, one series per segment, with each segment's scalar rho overlaid. + fig, ax = plt.subplots(figsize=(10, 5)) + for label, seg, color in segments: + daily = _daily_segment_ratio(wide.index, test_pw, ref_total, seg) + ax.plot(daily.index.to_numpy(), daily.to_numpy(), marker=".", linewidth=0.8, color=color, label=label) + rho = _rho(test_pw, ref_total, seg) + span = wide.index[seg] + if np.isfinite(rho) and len(span): + ax.hlines(rho, span.min(), span.max(), color=color, linestyle="--", linewidth=1.5) + ax.axvline(upgrade_start, color="k", linestyle="--", label="upgrade start") + ax.set_xlabel("date") + ax.set_ylabel("test / reference-total ratio") + ax.set_title(f"{test}: daily test/reference ratio (dashed = rho used by estimate)") + ax.grid(visible=True, alpha=0.3) + ax.legend() + fig.tight_layout() + _save(fig, plots_dir / stages.UPLIFT_RESULTS / f"{test}_ratio_timeseries.png") + + # 3) daily used-data coverage as a fraction of the day's expected timestamps, one series per + # segment, so each segment is seen to receive its share (under toggle, ~50% each post-upgrade). + expected_per_day = _expected_per_day(wide.index, timebase) + fig, ax = plt.subplots(figsize=(10, 5)) + for label, _seg, color in segments: + seg_mask = baseline_mask if label == "baseline" else toggle_rows.upgraded + daily = _daily_segment_coverage(wide.index, used, seg_mask, expected_per_day) + ax.plot(daily.index.to_numpy(), daily.to_numpy(), marker=".", linewidth=0.8, color=color, label=label) + ax.axvline(upgrade_start, color="k", linestyle="--", label="upgrade start") + ax.set_ylim(0.0, 1.0) + ax.yaxis.set_major_formatter(PercentFormatter(xmax=1.0)) + ax.set_xlabel("date") + ax.set_ylabel("used-data coverage") + ax.set_title(f"{test}: daily used-data coverage (complete-case, % of expected timestamps)") + ax.grid(visible=True, alpha=0.3) + ax.legend() + fig.tight_layout() + _save(fig, plots_dir / stages.FILTER / f"{test}_coverage_timeseries.png") + + +def _save(fig: plt.Figure, path: Path) -> None: + """Write a figure to ``path`` (creating its stage subfolder) and close it.""" + path.parent.mkdir(parents=True, exist_ok=True) + fig.savefig(path, dpi=150) + plt.close(fig) diff --git a/benchmarking/baselines/overnight_common.py b/benchmarking/baselines/overnight_common.py new file mode 100644 index 00000000..4600b0a0 --- /dev/null +++ b/benchmarking/baselines/overnight_common.py @@ -0,0 +1,71 @@ +"""Shared setup for the overnight prepost/toggle studies: a fresh per-run output dir and log. + +Each run gets its own timestamped output directory plus a ``run.log`` recording provenance (git +commit, study config, timings), so results from different runs never overwrite or +silently intermix (the older studies dumped everything into a single ``prepost``/``toggle`` dir, +leaving stale ``leaderboard_all_profiles.csv`` and accumulating per-run diagnostic CSVs), and every +output is traceable to the exact code that produced it. +""" + +from __future__ import annotations + +import logging +import subprocess +from datetime import datetime, timezone +from pathlib import Path +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from benchmarking.harness import StudyConfig + +logger = logging.getLogger(__name__) + +_REPO_DIR = Path(__file__).resolve().parent + + +def _git_commit() -> str: + """Return the current commit (with a ``-dirty`` suffix if the tree has uncommitted changes).""" + try: + commit = subprocess.run( + ["git", "rev-parse", "HEAD"], # noqa: S607 + cwd=_REPO_DIR, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + dirty = subprocess.run( + ["git", "status", "--porcelain"], # noqa: S607 + cwd=_REPO_DIR, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + except (subprocess.SubprocessError, OSError): + return "unknown" + return f"{commit}-dirty" if dirty else commit + + +def start_overnight_run(mode: str, study: StudyConfig, output_root: Path) -> Path: + """Create a fresh timestamped output dir for one overnight run and wire logging into it. + + :param mode: ``"prepost"`` or ``"toggle"`` (the per-mode subfolder) + :param study: the study configuration (logged for provenance) + :param output_root: the base ``WIND_UP_BENCHMARKING_OUTPUT_DIR`` (the studies' ``default_output_root`` + already appends ``mode``; pass its ``.parent`` so we can add the timestamp under ``mode``) + :return: the per-run directory to pass as ``out_root`` to the study runner + """ + timestamp = datetime.now(timezone.utc).strftime("%Y%m%d_%H%M%S") + out_dir = output_root / mode / timestamp + out_dir.mkdir(parents=True, exist_ok=True) + + logging.basicConfig( + level=logging.INFO, + format="%(asctime)s %(levelname)s %(name)s: %(message)s", + handlers=[logging.StreamHandler(), logging.FileHandler(out_dir / "run.log")], + force=True, + ) + logger.info("Starting %s overnight run", mode) + logger.info("git commit: %s", _git_commit()) + logger.info("output dir: %s", out_dir) + logger.info("study config: %s", study) + return out_dir diff --git a/benchmarking/baselines/overnight_profiles.py b/benchmarking/baselines/overnight_profiles.py new file mode 100644 index 00000000..997483fa --- /dev/null +++ b/benchmarking/baselines/overnight_profiles.py @@ -0,0 +1,35 @@ +"""The shared upgrade-profile set for the longer (overnight) prepost and toggle studies. + +Defined once so the prepost and toggle runs score an identical set. Seven profiles spanning +sign, magnitude and shape: + +* constant Cp: -10%, 0% (placebo), +3%, +10% +* wind-speed-dependent Cp: +10% held below 5 m/s, fading linearly to 0% by 12 m/s +* TI-dependent Cp: +10% held below 10% turbulence intensity, fading to 0% by 30% TI +* rated-power uprate: +5% (HoT rated = 2300 kW -> 2415 kW) +""" + +from __future__ import annotations + +from benchmarking.synthetic import ( + HOT_RATED_POWER_KW, + ConditionCpChange, + ConstantCpChange, + RatedPowerChange, + WindSpeedCpChange, +) + + +def overnight_profiles() -> dict[str, list]: + """Return the shared {name -> list of upgrade effects} mapping for the overnight studies.""" + return { + "cp_minus_10pct": [ConstantCpChange(delta=-0.10)], + "cp_0pct": [ConstantCpChange(delta=0.0)], + "cp_plus_3pct": [ConstantCpChange(delta=0.03)], + "cp_plus_10pct": [ConstantCpChange(delta=0.10)], + # +10% Cp held below 5 m/s, fading linearly to 0% by 12 m/s (endpoints held outside range) + "ws_dependent_cp": [WindSpeedCpChange(ws_points=(5.0, 12.0), deltas=(0.10, 0.0))], + # +10% Cp held below 10% TI, fading linearly to 0% by 30% TI (TI = ws_sd / ws, a fraction) + "ti_dependent_cp": [ConditionCpChange(by="ti", points=(0.10, 0.30), deltas=(0.10, 0.0))], + "rated_plus_5pct": [RatedPowerChange(new_rated_power_kw=HOT_RATED_POWER_KW * 1.05)], + } diff --git a/benchmarking/baselines/power_model/__init__.py b/benchmarking/baselines/power_model/__init__.py new file mode 100644 index 00000000..ce8db829 --- /dev/null +++ b/benchmarking/baselines/power_model/__init__.py @@ -0,0 +1,12 @@ +"""Counterfactual power-model uplift method (v1 — simplest-possible ML). + +A pluggable, v0-independent counterfactual power model behind the harness ``Method`` seam: learn +the test turbine's normal power from curated reference-only (weather + wake) features over the +baseline, predict the counterfactual over the upgraded window, and take the energy ratio. +""" + +from __future__ import annotations + +from benchmarking.baselines.power_model.method import CURATED_ERA5_EXCLUDE, TUNED_MODEL_PARAMS, PowerModelMethod + +__all__ = ["CURATED_ERA5_EXCLUDE", "TUNED_MODEL_PARAMS", "PowerModelMethod"] diff --git a/benchmarking/baselines/power_model/conditional.py b/benchmarking/baselines/power_model/conditional.py new file mode 100644 index 00000000..141d6ef3 --- /dev/null +++ b/benchmarking/baselines/power_model/conditional.py @@ -0,0 +1,81 @@ +"""Pure conditional-decomposition helpers (imputation + energy-identity re-level). + +Extracted from ``method.py`` so they are unit-testable without a fit and keep the method file +focused. The imputation fills the per-bin uplift *shape* for bins the two-direction combine could +not measure (a degenerate non-positive ratio, or too few matched rows once the count floor is on); +the re-level pins those imputed bins and rescales only the measured bins so the whole decomposition +energy-aggregates back to the headline exactly. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd + + +def impute_uncovered_bins( + one_plus_u: np.ndarray, + *, + condition: str, + measured: np.ndarray, + one_plus_overall: float, +) -> np.ndarray: + """Fill the per-bin ``1+u`` shape for uncovered bins; measured bins pass through unchanged. + + Bins **must** be in ascending bin order (low ws / low TI first). ``measured`` is the trust mask + (``True`` = keep the two-direction shape). Uncovered bins: + + - ``ws`` / ``power``: backward-fill from the nearest covered bin above (uplift is Cp-shaped and + monotone-saturating in both, so a low gap looks most like the next covered bin up), then any + bins above the last covered one take ``1.0`` — 0 uplift, since at rated both baseline and + upgraded hit rated power. This 0-at-rated prior is wrong for uprating / power-boost upgrades; it + is a documented, replaceable default. + - ``ti``: no ordering physics, so uncovered bins take the overall uplift (``one_plus_overall``). + + The result is always all-finite: ws/power fall back to ``1.0`` (even with no measured bins), ti to + ``one_plus_overall``. Raises for a ``condition`` other than ``"ws"`` / ``"ti"`` / ``"power"`` so a + mistyped axis fails loudly rather than silently taking the ti branch. + """ + if condition not in ("ws", "ti", "power"): + msg = f"unknown condition {condition!r}; expected 'ws', 'ti' or 'power'" + raise ValueError(msg) + s = pd.Series(np.asarray(one_plus_u, dtype=float)) + s[~np.asarray(measured, dtype=bool)] = np.nan + s = s.bfill().fillna(1.0) if condition in ("ws", "power") else s.fillna(float(one_plus_overall)) + return s.to_numpy() + + +def relevel_conditional( + sum_actual_b: np.ndarray, + one_plus_u_b: np.ndarray, + *, + measured: np.ndarray, + one_plus_overall: float, +) -> np.ndarray: + """Rescale measured bins by one λ (imputed bins pinned) so the decomposition aggregates to overall. + + The aggregation is the ratio-of-sums ``Σactual / Σ(actual/(1+u))`` and must equal + ``one_plus_overall``. Imputed bins (``~measured``) + contribute a fixed counterfactual energy ``C_i = Σ_imp actual/(1+u_imp)``; measured bins scale as + ``1+u -> λ(1+u)`` so their counterfactual energy is ``S_m/λ`` with ``S_m = Σ_meas actual/(1+u)``. + Setting ``S_m/λ + C_i = Σactual / one_plus_overall`` gives + ``λ = S_m / (Σactual/one_plus_overall - C_i)``. If there are no measured bins or the denominator is + non-positive (imputed energy already exceeds the headline total), λ cannot be solved for a positive + scale — fall back to reporting ``one_plus_overall`` in every bin. + """ + a = np.asarray(sum_actual_b, dtype=float) + u1 = np.asarray(one_plus_u_b, dtype=float) + is_measured = np.asarray(measured, dtype=bool) + usable = np.isfinite(u1) & (u1 != 0) & np.isfinite(a) + m = is_measured & usable + imp = (~is_measured) & usable + total_actual = float(a[np.isfinite(a)].sum()) + s_m = float((a[m] / u1[m]).sum()) if m.any() else 0.0 + c_i = float((a[imp] / u1[imp]).sum()) if imp.any() else 0.0 + denom = total_actual / one_plus_overall - c_i + if not m.any() or denom <= 0: + return np.full_like(u1, float(one_plus_overall)) + lam = s_m / denom + out = u1.copy() + out[m] = lam * u1[m] + return out diff --git a/benchmarking/baselines/power_model/diagnostics.py b/benchmarking/baselines/power_model/diagnostics.py new file mode 100644 index 00000000..e56b74bf --- /dev/null +++ b/benchmarking/baselines/power_model/diagnostics.py @@ -0,0 +1,707 @@ +"""Per-run diagnostics for the power model: CSVs, feature importance, and plots. + +A human reviewer must be able to confirm the right data was received and interpreted and — +critically — spot feature leakage (a feature that trivially predicts power rather than carrying +weather/wake information; design note §3). Hence the feature-importance table and plot are +first-class outputs and the top features are logged. The headline uplift is re-derivable from the +results CSV as ``sum_actual / sum_counterfactual - 1``. + +Everything here is pure reporting; the estimate itself is computed in :mod:`method`. +""" + +from __future__ import annotations + +import logging +import re +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any + +import matplotlib.pyplot as plt +import numpy as np +import numpy.typing as npt +import pandas as pd +from matplotlib.colors import Normalize +from matplotlib.patches import Patch + +from benchmarking.baselines.power_model.features import QUALIFIER +from benchmarking.diagnostics import stages +from benchmarking.diagnostics.density import density_scatter +from benchmarking.diagnostics.style import apply_grid, save_fig +from benchmarking.harness.conditions import CONDITIONS, TI_BINS, WS_BINS + +if TYPE_CHECKING: + from pathlib import Path + +logger = logging.getLogger(__name__) + +_SEGMENTS = ("all", "baseline", "upgraded") +_TOP_FEATURES_LOGGED = 12 +_MIN_CORR_PAIRS = 2 + + +@dataclass +class DiagnosticData: + """Everything the diagnostics need; assembled by :class:`~.method.PowerModelMethod`.""" + + test_wtg: str + mode: str + index: pd.DatetimeIndex # all unique timestamps + treated_all: np.ndarray # upgrade flag (0/1) over all timestamps + selected_all: np.ndarray # bool: normally-operating rows with finite power + y_all: np.ndarray # test power over all timestamps + timebase: pd.Timedelta + # counterfactual on the upgraded selected rows + upgraded_ts: pd.DatetimeIndex + y_upgraded: np.ndarray + pred_upgraded: np.ndarray # counterfactual expected power had there been no upgrade + # held-out baseline fit quality + y_baseline_valid: np.ndarray + pred_baseline_valid: np.ndarray + # feature catalogue / importance + feature_names: list[str] + feature_values: pd.DataFrame # selected feature matrix X (baseline+upgraded selected rows) + y_selected: np.ndarray # test power over those selected rows (for |corr|) + outcome_model: Any + overall_uplift: float + sum_actual_kw: float + sum_counterfactual_kw: float + n_refs: int + era5_lag_rows: int | None + era5_corr: float | None + era5_sweep: pd.DataFrame | None + # test-turbine ws/TI row-aligned to each segment's residuals (None when no wind-speed col) + cond_upgraded: pd.DataFrame | None = None + cond_baseline_valid: pd.DataFrame | None = None + + +def feature_importance_long(data: DiagnosticData) -> pd.DataFrame: + """Long table of LightGBM gain/split importance for the (single) outcome power model. + + An alternative learner injected via the model-factory seam has no ``booster_``; the table then + carries NaN importances (the feature *catalogue* still works) rather than failing the run. + """ + booster = getattr(data.outcome_model, "booster_", None) + if booster is None: + return pd.DataFrame({"feature": data.feature_names, "gain": np.nan, "split_count": np.nan}) + return pd.DataFrame( + { + "feature": data.feature_names, + "gain": booster.feature_importance(importance_type="gain"), + "split_count": booster.feature_importance(importance_type="split"), + } + ).sort_values("gain", ascending=False, ignore_index=True) + + +def log_top_features(importance: pd.DataFrame) -> None: + """Log the model's top features so a human can spot a leaking (too-good) predictor.""" + top = importance.head(_TOP_FEATURES_LOGGED) + pairs = ", ".join(f"{r.feature} (gain={r.gain:.0f})" for r in top.itertuples()) + logger.info("power_model top features by gain: %s", pairs) + logger.info( + "Review the above: weather + wake tags are expected; a feature that trivially predicts power is a flag." + ) + + +def segment_stats(data: DiagnosticData) -> pd.DataFrame: + """Per-segment (all/baseline/upgraded) counts and energy, for a human data sanity check.""" + timebase_hours = data.timebase / pd.Timedelta(hours=1) + treated = data.treated_all.astype(bool) + masks = {"all": np.ones(len(data.index), dtype=bool), "baseline": ~treated, "upgraded": treated} + rows = [] + for segment in _SEGMENTS: + seg = masks[segment] + seg_sel = seg & data.selected_all + seg_power = data.y_all[seg_sel] + finite = seg_power[np.isfinite(seg_power)] + rows.append( + { + "segment": segment, + "first_timestamp": data.index[seg].min() if seg.any() else pd.NaT, + "last_timestamp": data.index[seg].max() if seg.any() else pd.NaT, + "n_timestamps": int(seg.sum()), + "n_selected": int(seg_sel.sum()), + "selected_fraction": float(seg_sel.sum() / seg.sum()) if seg.any() else np.nan, + "test_mean_power_kw": float(finite.mean()) if len(finite) else np.nan, + "test_mwh": float(finite.sum()) * timebase_hours / 1000.0 if len(finite) else np.nan, + } + ) + return pd.DataFrame(rows) + + +def results_row(data: DiagnosticData) -> pd.DataFrame: + """Single-row headline results: uplift, energy totals, baseline fit quality, ERA5 sync.""" + timebase_hours = data.timebase / pd.Timedelta(hours=1) + r2, mae = _r2_mae(data.y_baseline_valid, data.pred_baseline_valid) + return pd.DataFrame( + [ + { + "test_wtg": data.test_wtg, + "mode": data.mode, + "n_refs": data.n_refs, + "n_timestamps": len(data.index), + "n_selected": int(data.selected_all.sum()), + "n_selected_upgraded": len(data.y_upgraded), + "n_features": len(data.feature_names), + "uplift_frc": data.overall_uplift, + "sum_actual_mwh": data.sum_actual_kw * timebase_hours / 1000.0, + "sum_counterfactual_mwh": data.sum_counterfactual_kw * timebase_hours / 1000.0, + "baseline_holdout_r2": r2, + "baseline_holdout_mae_kw": mae, + "era5_lag_rows": data.era5_lag_rows, + "era5_corr": data.era5_corr, + "time_calculated": pd.Timestamp.utcnow(), + } + ] + ) + + +def feature_catalogue(data: DiagnosticData) -> pd.DataFrame: + """One row per feature: source tag/turbine, coverage, basic stats, gain, |corr| with power.""" + importance = feature_importance_long(data).set_index("feature") + rows = [] + for feature in data.feature_names: + col = data.feature_values[feature].to_numpy(dtype=float) + finite = np.isfinite(col) + tag, _, turbine = feature.partition(QUALIFIER) + rows.append( + { + "feature": feature, + "source_tag": tag, + "turbine": turbine or "ERA5/derived", + "coverage_pct": float(100.0 * finite.mean()) if len(col) else np.nan, + "mean": float(np.nanmean(col)) if finite.any() else np.nan, + "std": float(np.nanstd(col)) if finite.any() else np.nan, + "min": float(np.nanmin(col)) if finite.any() else np.nan, + "max": float(np.nanmax(col)) if finite.any() else np.nan, + "gain": float(importance["gain"].get(feature, 0.0)), + "abs_corr_with_power": _abs_corr(col, data.y_selected), + } + ) + return pd.DataFrame(rows).sort_values("gain", ascending=False, ignore_index=True) + + +def _abs_corr(col: np.ndarray, y: np.ndarray) -> float: + """Absolute Pearson correlation of a feature with the outcome over their finite pairs.""" + pair = np.isfinite(col) & np.isfinite(y) + if pair.sum() < _MIN_CORR_PAIRS or np.std(col[pair]) == 0 or np.std(y[pair]) == 0: + return float("nan") + return float(abs(np.corrcoef(col[pair], y[pair])[0, 1])) + + +def _r2_mae(actual: np.ndarray, predicted: np.ndarray) -> tuple[float, float]: + """Return (R², MAE) over the finite pairs of ``actual`` / ``predicted``.""" + finite = np.isfinite(actual) & np.isfinite(predicted) + actual, predicted = actual[finite], predicted[finite] + if len(actual) == 0: + return float("nan"), float("nan") + resid = actual - predicted + ss_res = float(np.sum(resid**2)) + ss_tot = float(np.sum((actual - actual.mean()) ** 2)) + r2 = 1.0 - ss_res / ss_tot if ss_tot else float("nan") + return r2, float(np.mean(np.abs(resid))) + + +def write_csvs(run_dir: Path, run_name: str, ts: str, data: DiagnosticData) -> pd.DataFrame: + """Write the data-stats, results, feature-importance and feature-catalogue CSVs; return importance.""" + segment_stats(data).to_csv(run_dir / f"{run_name}_data_stats_{ts}.csv", index=False) + results_row(data).to_csv(run_dir / f"{run_name}_results_{ts}.csv", index=False) + importance = feature_importance_long(data) + importance.to_csv(run_dir / f"{run_name}_feature_importance_{ts}.csv", index=False) + feature_catalogue(data).to_csv(run_dir / f"{run_name}_feature_catalogue_{ts}.csv", index=False) + return importance + + +def write_conditional_csvs( + run_dir: Path, + run_name: str, + ts: str, + *, + overall: dict[str, Any], + per_bin: pd.DataFrame | None, + match: Any, # noqa: ANN401 - MatchResult (avoids importing the matching module into diagnostics) +) -> None: + """Write the conditional-step diagnostics: implied shrinkage ``s`` + the CEM balance. + + ``overall`` is the single-row headline (both directions' ratios, the headline uplift, and the + implied shrinkage the two directions cancelled); ``per_bin`` the same per (ws, TI) bin; ``match`` the + CEM balance/coverage (retained fractions, effective sample size, one-sided cells dropped) plus the + per-cell counts. Together they show, for one case, how much conditional shrinkage was cancelled and + how healthy the matching was. ``run_dir`` is the run's ``conditional/`` subfolder. + """ + pd.DataFrame([overall]).to_csv(run_dir / f"{run_name}_conditional_overall_{ts}.csv", index=False) + if per_bin is not None: + per_bin.to_csv(run_dir / f"{run_name}_conditional_by_bin_{ts}.csv", index=False) + balance = { + "n_baseline_in": match.n_baseline_in, + "n_upgraded_in": match.n_upgraded_in, + "n_matched_per_side": match.n_matched_per_side, + "retained_fraction_baseline": match.retained_fraction_baseline, + "retained_fraction_upgraded": match.retained_fraction_upgraded, + "n_cells_two_sided": match.n_cells_two_sided, + "n_cells_one_sided": match.n_cells_one_sided, + } + pd.DataFrame([balance]).to_csv(run_dir / f"{run_name}_cem_balance_{ts}.csv", index=False) + match.per_cell.to_csv(run_dir / f"{run_name}_cem_cells_{ts}.csv", index=False) + + +# covered (measured two-direction shape) vs imputed bins on the uplift panel. +_COVERED_COLOR = "C0" +_IMPUTED_COLOR = "0.7" + + +def _condition_diagnostic_figure(sub: pd.DataFrame, *, condition: str, test_wtg: str) -> plt.Figure: + """Build a four-panel per-condition diagnostic: forward + reverse measurement, uplift, implied shrinkage. + + ``sub`` is the per-bin rows for one condition (ascending bin order). The uplift panel shades each bar + by ``covered``: a measured (two-direction) bin vs an imputed one, so it is clear which per-bin uplifts + are backed by data. Implied shrinkage carries the ``s = 1`` reference (agreement / nothing to cancel). + """ + labels = sub["condition_bin"].astype(str).to_numpy() + covered = sub["covered"].to_numpy(dtype=bool) + fig, axes = plt.subplots(2, 2, figsize=(13, 9)) + + def _bars( + ax: plt.Axes, + values: npt.ArrayLike, + *, + title: str, + ylabel: str, + color: Any, # noqa: ANN401 + baseline: float, + baseline_label: str | None = None, + ) -> None: + ax.bar(labels, np.asarray(values, dtype=float), color=color) + ax.axhline(baseline, color="k", linewidth=1, linestyle="--", label=baseline_label) + ax.set_title(title) + ax.set_xlabel(f"{condition} bin") + ax.set_ylabel(ylabel) + ax.tick_params(axis="x", labelrotation=90) + apply_grid(ax) + + _bars( + axes[0][0], + sub["r_fwd"], + title="forward measurement", + ylabel="forward energy ratio r_fwd", + color="C0", + baseline=0.0, + ) + _bars( + axes[0][1], + sub["r_rev"], + title="reverse measurement", + ylabel="reverse energy ratio r_rev", + color="C1", + baseline=0.0, + ) + + # uplift panel: shade covered (measured) vs imputed bins. + uplift_colors = [_COVERED_COLOR if c else _IMPUTED_COLOR for c in covered] + _bars( + axes[1][0], + sub["p50_uplift"], + title="uplift (measured / imputed)", + ylabel="p50 uplift", + color=uplift_colors, + baseline=0.0, + ) + handles = [ + Patch(facecolor=_COVERED_COLOR, label="measured (covered)"), + Patch(facecolor=_IMPUTED_COLOR, label="imputed"), + ] + axes[1][0].legend(handles=handles, loc="best") + + _bars( + axes[1][1], + sub["implied_shrinkage"], + title="implied shrinkage cancelled", + ylabel="implied shrinkage s", + color="C2", + baseline=1.0, + baseline_label="s = 1 (no shrinkage)", + ) + axes[1][1].legend(loc="lower right") + + fig.suptitle(f"{test_wtg} — {condition}: conditional two-direction diagnostics") + fig.tight_layout() + return fig + + +def plot_conditional_diagnostics(plots_dir: Path, per_bin: pd.DataFrame, *, test_wtg: str) -> None: + """Write one four-panel diagnostic figure per condition (ws, ti, power) present in ``per_bin``.""" + plots_dir.mkdir(parents=True, exist_ok=True) + present = set(per_bin["condition"]) + for cond in [c for c in CONDITIONS if c in present]: + sub = per_bin[per_bin["condition"] == cond] + fig = _condition_diagnostic_figure(sub, condition=cond, test_wtg=test_wtg) + save_fig(fig, plots_dir / f"conditional_{cond}.png") + + +def save_plots(plots_dir: Path, data: DiagnosticData, importance: pd.DataFrame) -> None: + """Write the power-model diagnostic plots into their analysis-stage subfolders.""" + model_dir = plots_dir / stages.UPLIFT_MODELLING + model_dir.mkdir(parents=True, exist_ok=True) + _plot_importance(model_dir, importance, test_wtg=data.test_wtg) + _plot_predicted_vs_actual(model_dir, data) + _plot_residual_vs_mean(model_dir, data) + _plot_residual_binned(model_dir, data) + + results_dir = plots_dir / stages.UPLIFT_RESULTS + results_dir.mkdir(parents=True, exist_ok=True) + _plot_actual_vs_counterfactual_timeseries(results_dir, data) + + inputs_dir = plots_dir / stages.UPLIFT_INPUTS + inputs_dir.mkdir(parents=True, exist_ok=True) + _plot_feature_overview(inputs_dir, feature_catalogue(data)) + _save_feature_histograms(inputs_dir / "feature_histograms", data) + + if data.era5_sweep is not None: + feat_dir = plots_dir / stages.FEATURE_ENG + feat_dir.mkdir(parents=True, exist_ok=True) + _plot_era5_sweep(feat_dir, data) + + +def _plot_importance(plots_dir: Path, importance: pd.DataFrame, *, test_wtg: str) -> None: + """Top features by gain for the single power model — the leakage / feature-thesis check.""" + top = importance.head(20).iloc[::-1] + fig, ax = plt.subplots(figsize=(11, max(6.0, 0.4 * len(top)))) + ax.barh(top["feature"], top["gain"], color="C0") + ax.set_xlabel("gain") + ax.set_title(f"{test_wtg}: power-model feature importance — expect weather + wake tags to dominate") + apply_grid(ax) + save_fig(fig, plots_dir / "feature_importance.png") + + +def _plot_feature_overview(plots_dir: Path, catalogue: pd.DataFrame) -> None: + """All features as a horizontal bar of gain, coloured by coverage — the add/remove overview.""" + df = catalogue.sort_values("gain", ascending=True) + norm = Normalize(vmin=0.0, vmax=100.0) + cmap = plt.get_cmap("viridis") + fig, ax = plt.subplots(figsize=(12, max(6.0, 0.3 * len(df)))) + ax.barh(df["feature"], df["gain"], color=cmap(norm(df["coverage_pct"].to_numpy()))) + fig.colorbar(plt.cm.ScalarMappable(norm=norm, cmap=cmap), ax=ax, label="coverage [%]") + ax.set_xlabel("outcome-model gain") + ax.set_title("all features: importance (bar) and coverage (colour) — short + cold = drop candidate") + apply_grid(ax) + save_fig(fig, plots_dir / "feature_overview.png") + + +def _save_feature_histograms(out_dir: Path, data: DiagnosticData) -> None: + """One baseline-vs-upgraded density histogram per model input feature, using the real tag name. + + Every column the uplift model actually sees gets its own plot (named by its real source tag / + ERA5 column) so a reviewer can check the baseline and upgraded distributions overlap for *each* + input — the per-feature view of the overlap/positivity story, complementing the curated shared + ``condition_histograms.png``. + """ + out_dir.mkdir(parents=True, exist_ok=True) + sel = np.asarray(data.selected_all, dtype=bool) + treated_sel = np.asarray(data.treated_all, dtype=bool)[sel] # aligned to feature_values rows + baseline_sel = ~treated_sel + for feature in data.feature_names: + vals = data.feature_values[feature].to_numpy(dtype=float) + bins = _robust_bins(vals) + fig, ax = plt.subplots(figsize=(7, 5)) + for seg_label, seg, color in (("baseline", baseline_sel, "C0"), ("upgraded", treated_sel, "C1")): + seg_vals = vals[seg & np.isfinite(vals)] + if seg_vals.size: + ax.hist( + seg_vals, bins=bins, density=True, histtype="stepfilled", alpha=0.45, color=color, label=seg_label + ) + ax.set_xlabel(feature) + ax.set_ylabel("density") + ax.set_title(f"{data.test_wtg}: {feature}\n(used rows, baseline vs upgraded)") + apply_grid(ax) + if ax.get_legend_handles_labels()[0]: + ax.legend() + save_fig(fig, out_dir / f"{_safe_filename(feature)}.png") + + +def _robust_bins(values: np.ndarray, *, bins: int = 30) -> list[float] | int: + """Bin edges over the 1st-99th percentile of the finite values, so outliers don't dominate.""" + finite = values[np.isfinite(values)] + if finite.size == 0: + return bins + lo, hi = np.percentile(finite, [1, 99]) + if hi <= lo: + return bins + return np.linspace(lo, hi, bins + 1).tolist() + + +def _safe_filename(name: str) -> str: + """Turn a feature name (which may contain ``@``, spaces, ``/``) into a safe file stem.""" + return re.sub(r"[^0-9A-Za-z._-]+", "_", name).strip("_") + + +def _plot_predicted_vs_actual(plots_dir: Path, data: DiagnosticData) -> None: + """Two panels: held-out baseline fit (should hug 1:1) and the upgraded counterfactual gap. + + On the upgraded panel the points sit *above* the 1:1 line by roughly the uplift: actual upgraded + power exceeds the counterfactual the model predicts from the (upgrade-blind) references. + """ + fig, axes = plt.subplots(1, 2, figsize=(15, 6.5)) + + yb, pb = data.y_baseline_valid, data.pred_baseline_valid + lim_b = [0.0, float(np.nanmax(yb))] if len(yb) else [0.0, 1.0] + density_scatter(yb, pb, ax=axes[0], s=6, colorbar=True) + axes[0].plot(lim_b, lim_b, color="red", linewidth=1.2, label="1:1") + r2, mae = _r2_mae(yb, pb) + axes[0].set_title(f"baseline (held-out) R²={r2:.3f}, MAE={mae:.0f} kW, n={len(yb)}") + axes[0].set_xlabel("actual power [kW]") + axes[0].set_ylabel("predicted power [kW]") + axes[0].legend(loc="upper left") + apply_grid(axes[0]) + + yu, pu = data.y_upgraded, data.pred_upgraded + lim_u = [0.0, float(np.nanmax(yu))] if len(yu) else [0.0, 1.0] + density_scatter(yu, pu, ax=axes[1], s=6, colorbar=True) + axes[1].plot(lim_u, lim_u, color="red", linewidth=1.2, label="1:1 (no uplift)") + axes[1].set_title(f"upgraded: actual vs counterfactual uplift={100 * data.overall_uplift:+.2f}%, n={len(yu)}") + axes[1].set_xlabel("actual power [kW]") + axes[1].set_ylabel("counterfactual predicted power [kW]") + axes[1].legend(loc="upper left") + apply_grid(axes[1]) + + fig.suptitle(f"{data.test_wtg}: power-model predicted vs actual") + save_fig(fig, plots_dir / "predicted_vs_actual.png") + + +def _plot_residual_vs_mean(plots_dir: Path, data: DiagnosticData) -> None: + """Bland-Altman: residual (actual - predicted) vs the mean of predicted and actual power. + + Two panels (held-out baseline, upgraded). The baseline residuals should sit on zero with no + trend across the power range (a tilt would flag a conditional bias in the fit). On the upgraded + panel the residual *is* the uplift signal, so it sits above zero by the mean uplift in kW + (annotated) — a flat band is a clean additive uplift; a slope means the uplift varies with + power level. + """ + fig, axes = plt.subplots(1, 2, figsize=(15, 6.5), sharey=True) + panels = ( + ("baseline (held-out)", data.y_baseline_valid, data.pred_baseline_valid), + ("upgraded", data.y_upgraded, data.pred_upgraded), + ) + for ax, (label, actual, predicted) in zip(axes, panels, strict=True): + resid = actual - predicted + mean_power = 0.5 * (actual + predicted) + finite = np.isfinite(resid) & np.isfinite(mean_power) + density_scatter(mean_power[finite], resid[finite], ax=ax, s=6, colorbar=(label == "upgraded")) + ax.axhline(0, color="k", linewidth=1) + mean_resid = float(np.mean(resid[finite])) if finite.any() else float("nan") + ax.axhline( + mean_resid, color="red", linestyle="--", linewidth=1.2, label=f"mean residual = {mean_resid:+.0f} kW" + ) + ax.set_title(f"{label} (n={int(finite.sum())})") + ax.set_xlabel("mean of predicted and actual power [kW]") + ax.set_ylabel("residual (actual - predicted) [kW]") + ax.legend(loc="upper left") + apply_grid(ax) + fig.suptitle(f"{data.test_wtg}: power-model residual vs mean power (Bland-Altman)") + save_fig(fig, plots_dir / "residual_vs_mean.png") + + +_RESID_POWER_BINS = 20 +_MIN_BIN_COUNT = 3 # below this a bin's mean/SD is too noisy to plot + + +def _binned_stats( + x: npt.ArrayLike, y: npt.ArrayLike, edges: npt.ArrayLike +) -> tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray]: + """Bin ``y`` by ``x`` over ``edges``; return (bin centres, mean, SD, count), NaN-safe. + + Empty or sparsely populated bins (< :data:`_MIN_BIN_COUNT`) yield NaN mean/SD so a noisy tail + does not draw a misleading spike. SD is the sample standard deviation of the residual in the bin. + """ + x = np.asarray(x, dtype=float) + y = np.asarray(y, dtype=float) + edges = np.asarray(edges, dtype=float) + centers = 0.5 * (edges[:-1] + edges[1:]) + mean = np.full(len(centers), np.nan) + sd = np.full(len(centers), np.nan) + count = np.zeros(len(centers), dtype=int) + finite = np.isfinite(x) & np.isfinite(y) + if finite.any(): + cats = pd.cut(x[finite], bins=edges) + grouped = pd.Series(y[finite]).groupby(cats, observed=False) + count = grouped.size().to_numpy() + mean = grouped.mean().to_numpy() + sd = grouped.std().to_numpy() # ddof=1; NaN for singleton bins + thin = count < _MIN_BIN_COUNT + mean[thin] = np.nan + sd[thin] = np.nan + return centers, mean, sd, count + + +def _power_bin_edges(*arrays: np.ndarray | None, n_bins: int = _RESID_POWER_BINS) -> np.ndarray: + """Equal-width power-bin edges from 0 to the robust (99th-pct) max across the given arrays.""" + present = [np.asarray(a, dtype=float) for a in arrays if a is not None and len(a)] + combined = np.concatenate(present) if present else np.array([]) + finite = combined[np.isfinite(combined)] + hi = float(np.nanpercentile(finite, 99)) if finite.size else 1.0 + if not np.isfinite(hi) or hi <= 0: + hi = 1.0 + return np.linspace(0.0, hi, n_bins + 1) + + +def _axis_values(kind: str, actual: np.ndarray, predicted: np.ndarray, cond: pd.DataFrame | None) -> np.ndarray | None: + """Binning-axis values for one segment: power axes from actual/predicted, ws/TI from ``cond``.""" + if kind == "actual": + return np.asarray(actual, dtype=float) + if kind == "mean": + return 0.5 * (np.asarray(actual, dtype=float) + np.asarray(predicted, dtype=float)) + if cond is not None and kind in cond.columns: + return cond[kind].to_numpy(dtype=float) + return None + + +def _plot_residual_binned(plots_dir: Path, data: DiagnosticData) -> None: + """Mean and SD of the residual (actual - predicted) binned by power, wind speed and TI. + + The shrinkage check: a regularised learner predicts smoother than reality (over-predicts where + power is low, under-predicts where it is high), so on the **held-out baseline** — where the true + uplift is zero and the residual is pure model error — the mean residual tilts up with power. + Because power maps to wind speed, and TI is inversely related to wind speed at fixed power, that + single compression re-appears as a negative residual at low wind speed / high TI, which is what + masquerades as condition-dependent uplift. The ``upgraded`` segment adds the true uplift on top. + + Binning by *actual* power inflates the trend (the residual contains ``+actual``); the + ``mean(actual, predicted)`` (Bland-Altman) axis is the unbiased read. Both are shown so the + inflation is visible. + + Two files are written: ``residual_binned.png`` in absolute kW, and ``residual_binned_pct.png`` + where each bin's mean/SD residual is divided by that same bin's mean actual power (a true + percentage of the typical power in the bin, whatever the x-variable); bins with mean power <= 0 + are dropped. + """ + _residual_binned_figure(plots_dir / "residual_binned.png", data, as_percent=False) + _residual_binned_figure(plots_dir / "residual_binned_pct.png", data, as_percent=True) + + +def _residual_binned_figure(save_path: Path, data: DiagnosticData, *, as_percent: bool) -> None: + """Render one binned-residual figure (absolute kW or percentage of the mean x-variable).""" + segments = [ + ("baseline (held-out)", data.y_baseline_valid, data.pred_baseline_valid, data.cond_baseline_valid, "C0"), + ("upgraded", data.y_upgraded, data.pred_upgraded, data.cond_upgraded, "C1"), + ] + power_edges = _power_bin_edges(data.y_baseline_valid, data.pred_baseline_valid, data.y_upgraded, data.pred_upgraded) + # (axis kind, bin edges, x-axis label); ``_axis_values`` resolves the kind per segment + specs: list[tuple[str, np.ndarray, str]] = [ + ("actual", power_edges, "actual power [kW]"), + ("mean", power_edges, "mean(actual, predicted) [kW]"), + ("ws", np.asarray(WS_BINS, dtype=float), "wind speed [m/s]"), + ("ti", np.asarray(TI_BINS, dtype=float), "TI"), + ] + # keep an axis only if at least one segment yields values for it (ws/TI need a wind-speed col) + specs = [s for s in specs if any(_axis_values(s[0], a, p, c) is not None for _, a, p, c, _ in segments)] + if not specs: + return + + unit = "%" if as_percent else "kW" + fig, axes = plt.subplots(2, len(specs), figsize=(5.0 * len(specs), 8.5), squeeze=False, sharey="row") + mean_vals: list[np.ndarray] = [] # every plotted point, to size the shared y-axis from inliers + sd_vals: list[np.ndarray] = [] + for col, (kind, edges, xlabel) in enumerate(specs): + mean_ax, sd_ax = axes[0][col], axes[1][col] + for label, actual, predicted, cond, color in segments: + values = _axis_values(kind, actual, predicted, cond) + if values is None: + continue + actual_arr = np.asarray(actual, dtype=float) + resid = actual_arr - np.asarray(predicted, dtype=float) + centers, mean, sd, _ = _binned_stats(values, resid, edges) + if as_percent: + _, mean_power, _, _ = _binned_stats(values, actual_arr, edges) # per-bin denominator + mean = _as_percent_of_power(mean, mean_power) + sd = _as_percent_of_power(sd, mean_power) + mean_ax.plot(centers, mean, marker="o", ms=3, color=color, label=label) + sd_ax.plot(centers, sd, marker="o", ms=3, color=color, label=label) + mean_vals.append(mean) + sd_vals.append(sd) + mean_ax.axhline(0, color="k", lw=1) + mean_ax.set_title(xlabel) + mean_ax.set_ylabel(f"mean residual [{unit}]") + sd_ax.set_ylabel(f"SD of residual [{unit}]") + sd_ax.set_xlabel(xlabel) + for ax in (mean_ax, sd_ax): + ax.legend(loc="best", fontsize=8) + apply_grid(ax) + if as_percent: + # Size the shared y-axis from bins within ±30%; extreme low-power bins still plot but clip. + _set_ylim_from_inliers(axes[0][0], mean_vals) + _set_ylim_from_inliers(axes[1][0], sd_vals) + suffix = " (% of bin mean power)" if as_percent else "" + fig.suptitle(f"{data.test_wtg}: residual (actual - predicted) binned — shrinkage check{suffix}") + save_fig(fig, save_path) + + +_INLIER_PCT = 30.0 # bins beyond ±this (% of power) don't get to blow up the shared y-axis + + +def _set_ylim_from_inliers(ax: plt.Axes, value_arrays: list[np.ndarray]) -> None: + """Set ``ax`` y-limits from points within ±:data:`_INLIER_PCT`, with a small margin. + + Outliers (e.g. tiny-power bins with huge % residuals) are still drawn but fall outside the + limits and clip, so the readable bulk is not crushed. No-op if there are no inliers. + """ + if not value_arrays: + return + pooled = np.concatenate(value_arrays) + inliers = pooled[np.isfinite(pooled) & (np.abs(pooled) <= _INLIER_PCT)] + if inliers.size == 0: + return + lo, hi = float(inliers.min()), float(inliers.max()) + margin = 0.05 * (hi - lo) if hi > lo else max(abs(hi), 1.0) * 0.05 + ax.set_ylim(lo - margin, hi + margin) + + +def _as_percent_of_power(stat: npt.ArrayLike, mean_power: npt.ArrayLike) -> np.ndarray: + """Express a per-bin kW statistic as a percentage of that bin's mean power (NaN where <= 0).""" + power = np.asarray(mean_power, dtype=float) + denom = np.where(power > 0, power, np.nan) + return 100.0 * np.asarray(stat, dtype=float) / denom + + +def _plot_actual_vs_counterfactual_timeseries(plots_dir: Path, data: DiagnosticData) -> None: + """Daily energy: actual upgraded power vs the model's counterfactual, over the upgraded window.""" + timebase_hours = data.timebase / pd.Timedelta(hours=1) + actual = pd.Series(data.y_upgraded * timebase_hours / 1000.0, index=data.upgraded_ts) + counter = pd.Series(data.pred_upgraded * timebase_hours / 1000.0, index=data.upgraded_ts) + daily_actual = actual.resample("1D").sum(min_count=1) + daily_counter = counter.resample("1D").sum(min_count=1) + fig, ax = plt.subplots(figsize=(11, 5)) + ax.plot(daily_actual.index.to_numpy(), daily_actual.to_numpy(), marker=".", color="C1", label="actual (upgraded)") + ax.plot( + daily_counter.index.to_numpy(), + daily_counter.to_numpy(), + marker=".", + color="C0", + label="counterfactual (model)", + ) + ax.set_xlabel("date") + ax.set_ylabel("daily energy [MWh]") + ax.set_title(f"{data.test_wtg}: actual vs counterfactual daily energy (gap = uplift)") + apply_grid(ax) + ax.legend() + save_fig(fig, plots_dir / "actual_vs_counterfactual_timeseries.png") + + +def _plot_era5_sweep(plots_dir: Path, data: DiagnosticData) -> None: + """ERA5 correlation-vs-lag sweep, with the chosen optimal shift annotated.""" + sweep = data.era5_sweep + if sweep is None: + return + fig, ax = plt.subplots(figsize=(8, 6)) + ax.plot(sweep["shift_rows"], sweep["corr"], marker=".") + if data.era5_lag_rows is not None: + corr_text = f"{data.era5_corr:.3f}" if data.era5_corr is not None else "n/a" + ax.axvline( + data.era5_lag_rows, + color="k", + linestyle="--", + label=f"best shift = {data.era5_lag_rows} rows (corr = {corr_text})", + ) + ax.legend() + ax.set_xlabel("ERA5 shift [rows]") + ax.set_ylabel("wind-speed correlation") + ax.set_title(f"{data.test_wtg}: ERA5-SCADA correlation vs lag") + apply_grid(ax) + save_fig(fig, plots_dir / "era5_sync.png") diff --git a/benchmarking/baselines/power_model/features.py b/benchmarking/baselines/power_model/features.py new file mode 100644 index 00000000..de3a943e --- /dev/null +++ b/benchmarking/baselines/power_model/features.py @@ -0,0 +1,186 @@ +"""Build the power-model's curated, reference-only feature matrix. + +The discipline (design note §3): every model feature must be *upgrade-invariant* — derived from +reference turbines (or ERA5), never the test turbine's own signals, which the upgrade distorts. +Unlike the R-learner's *maximal* feature builder, this matrix is deliberately **curated** to +features known to relate to the *cause* of the test turbine's power — weather and wakes: + +* per **reference turbine**: active power (the primary stable weather-driven measurement) and the + availability counter (whether the reference is operating, hence whether it is making a wake); +* all raw **ERA5** columns, passed through under their original Open-Meteo names (no renaming), + with derived ``sin``/``cos`` companions for the circular wind-direction fields. + +Features that are not expected to add value and risk the model learning coincidences rather than +cause-effect (reactive power, blade pitch, …) are intentionally excluded. + +Feature columns from references are named ``"{QUALIFIER}"`` so the original tag is +preserved verbatim in importance diagnostics. :func:`check_reference_only` rejects any +test-turbine-qualified column (the §3 guard). +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd + +from benchmarking.baselines.era5_sync import ERA5_WD, ERA5_WS + +if TYPE_CHECKING: + from collections.abc import Sequence + +# Separator between a source-native tag and the turbine it came from in a feature name. +QUALIFIER = " @ " + + +def _references(scada_df: pd.DataFrame, *, test_wtg: str, turbine_col: str) -> list[str]: + """Sorted reference turbine names (every turbine present except the test turbine).""" + refs = sorted(t for t in scada_df[turbine_col].unique() if t != test_wtg) + if not refs: + msg = ( + f"no reference turbines available for test_wtg {test_wtg!r}: scada_df contains only " + f"{sorted(scada_df[turbine_col].unique())}. The power model needs at least one reference turbine." + ) + raise ValueError(msg) + return refs + + +def build_reference_features( + scada_df: pd.DataFrame, + *, + test_wtg: str, + turbine_col: str, + active_power_col: str, + availability_col: str, + extra_cols: Sequence[str] = (), + include_availability: bool = True, +) -> pd.DataFrame: + """Wide, curated reference features: each reference turbine's active power (+ optional extras). + + By default each reference contributes its active power and availability; ``extra_cols`` adds + more per-reference channels and ``include_availability=False`` drops the availability feature. + Columns are ``"{QUALIFIER}"`` keeping the original tag name. The test turbine + contributes nothing (its power is the outcome, extracted separately). NaNs are preserved (no + complete-case dropping) — LightGBM handles them natively. Raises if no reference turbine is + present, or (defensively) if any test-turbine column would leak in. + + :param extra_cols: additional per-reference value columns to carry as features (Issue 11's + active-power max/min/SD statistics); must be present in ``scada_df`` like the primary two + :param include_availability: when ``False``, drop the per-reference availability *feature* + (removal-ablation knob); ``availability_col`` must still exist in ``scada_df`` (the + downtime filter needs it), so its presence is validated either way + """ + refs = _references(scada_df, test_wtg=test_wtg, turbine_col=turbine_col) + value_cols = [active_power_col, *([availability_col] if include_availability else []), *extra_cols] + # availability_col stays validated even when not featured: it is a required input and the + # docstring contract is that it exists for the downstream downtime filter. + missing = sorted(c for c in {*value_cols, availability_col} if c not in scada_df.columns) + if missing: + msg = f"scada_df is missing required reference-feature columns {missing}; have {list(scada_df.columns)}" + raise ValueError(msg) + index = pd.DatetimeIndex(pd.unique(scada_df.index)).sort_values() + + tmp = scada_df[[turbine_col, *value_cols]].copy() + tmp["_ts"] = scada_df.index + wide = tmp.pivot_table(index="_ts", columns=turbine_col, values=value_cols, aggfunc="first") + keep = [(col, r) for col in value_cols for r in refs if (col, r) in wide.columns] + features = wide.loc[:, keep] + features.columns = [f"{col}{QUALIFIER}{r}" for col, r in keep] + features = features.reindex(index) + features.index.name = index.name + check_reference_only(features.columns.tolist(), test_wtg=test_wtg) + return features + + +def era5_feature_frame(aligned_era5: pd.DataFrame) -> pd.DataFrame: + """Turn aligned ERA5 into model features: all raw columns passed through + dir sin/cos companions. + + All raw Open-Meteo columns are kept under their original names (no renaming); the neutral + ``era5_ws`` / ``era5_wd`` aliases the sync adds for back-compat are dropped here (they duplicate + ``wind_speed_100m`` / ``wind_direction_100m``). Circular wind-direction fields additionally get + derived ``_sin`` / ``_cos`` companions (LightGBM cannot see that 359° ≈ 1°); the raw + degree columns are kept too. + """ + raw_cols = [c for c in aligned_era5.columns if c not in (ERA5_WS, ERA5_WD)] + out = aligned_era5[raw_cols].astype(float).copy() + for col in raw_cols: + if "direction" in col: + rad = np.deg2rad(out[col].to_numpy(dtype=float)) + out[f"{col}_sin"] = np.sin(rad) + out[f"{col}_cos"] = np.cos(rad) + return out + + +def extract_outcome( + scada_df: pd.DataFrame, + *, + test_wtg: str, + turbine_col: str, + active_power_col: str, +) -> pd.Series: + """Return the outcome ``y`` (the test turbine's active power) on the unique sorted index.""" + index = pd.DatetimeIndex(pd.unique(scada_df.index)).sort_values() + test_rows = scada_df[scada_df[turbine_col] == test_wtg] + y = pd.Series(test_rows[active_power_col].to_numpy(dtype=float), index=pd.DatetimeIndex(test_rows.index)) + return y[~y.index.duplicated()].reindex(index) + + +def reference_mean_wind_speed( + scada_df: pd.DataFrame, + *, + test_wtg: str, + turbine_col: str, + wind_speed_col: str, +) -> pd.Series: + """Mean wind speed across reference turbines on the unique index (used only for ERA5 lag sync). + + This is **not** a model feature — it is the site wind-speed signal the ERA5 correlation sweep + locks onto. Computed from references only so it stays upgrade-invariant. + """ + index = pd.DatetimeIndex(pd.unique(scada_df.index)).sort_values() + refs = _references(scada_df, test_wtg=test_wtg, turbine_col=turbine_col) + if wind_speed_col not in scada_df.columns: + return pd.Series(np.nan, index=index) + cols = [] + for r in refs: + rows = scada_df[scada_df[turbine_col] == r] + series = pd.Series(rows[wind_speed_col].to_numpy(dtype=float), index=pd.DatetimeIndex(rows.index)) + cols.append(series[~series.index.duplicated()].reindex(index)) + return pd.concat(cols, axis=1).mean(axis=1) + + +def test_condition_signals( + scada_df: pd.DataFrame, + *, + test_wtg: str, + turbine_col: str, + wind_speed_col: str, + wind_speed_sd_col: str | None, +) -> pd.DataFrame: + """Test turbine's MEASURED ws and ti on the unique sorted index (post-treatment, accepted §3). + + ``ti`` is omitted when no SD column is configured. + """ + index = pd.DatetimeIndex(pd.unique(scada_df.index)).sort_values() + rows = scada_df[scada_df[turbine_col] == test_wtg] + ws = pd.Series(rows[wind_speed_col].to_numpy(dtype=float), index=pd.DatetimeIndex(rows.index)) + ws = ws[~ws.index.duplicated()].reindex(index) + out = pd.DataFrame({"ws": ws}) + if wind_speed_sd_col is not None and wind_speed_sd_col in scada_df.columns: + sd = pd.Series(rows[wind_speed_sd_col].to_numpy(dtype=float), index=pd.DatetimeIndex(rows.index)) + sd = sd[~sd.index.duplicated()].reindex(index) + ws_arr = ws.to_numpy() + out["ti"] = np.divide(sd.to_numpy(), ws_arr, out=np.full(len(ws_arr), np.nan), where=ws_arr != 0) + return out + + +def check_reference_only(feature_names: list[str], *, test_wtg: str) -> None: + """Raise if any feature is qualified with the test turbine (violating the §3 rule).""" + offenders = [f for f in feature_names if f.endswith(f"{QUALIFIER}{test_wtg}")] + if offenders: + msg = ( + f"reference-only rule violated: features derived from the test turbine {test_wtg!r} " + f"are not allowed (the upgrade distorts its signals, design note §3): {offenders}" + ) + raise ValueError(msg) diff --git a/benchmarking/baselines/power_model/fitting.py b/benchmarking/baselines/power_model/fitting.py new file mode 100644 index 00000000..75c03616 --- /dev/null +++ b/benchmarking/baselines/power_model/fitting.py @@ -0,0 +1,27 @@ +"""Time-blocked fold assignment for the power model's baseline holdout fit. + +:func:`time_block_folds` — contiguous-block, round-robin fold assignment for time-ordered rows. +A shuffled split leaks autocorrelation (held-out rows sit minutes from training rows), so its +residuals are optimistic; contiguous blocks confine the leakage to the block edges, which makes +the held-out fit-quality diagnostic honest. +""" + +from __future__ import annotations + +import numpy as np + + +def time_block_folds(n: int, *, n_folds: int = 5, n_blocks: int = 25) -> np.ndarray: + """Assign ``n`` time-ordered rows to ``n_folds`` folds as round-robin contiguous blocks. + + Rows must already be in time order (the power model's row arrays are — they follow the sorted + analysis index). The rows are cut into ``n_blocks`` contiguous, equal-length blocks and block + ``i`` goes to fold ``i % n_folds``, so every fold samples all seasons while staying contiguous + at the scale that matters for autocorrelation (only the block edges sit near training rows). + Returns an int array of fold ids, one per row. + """ + if n_folds < 2 or n_blocks < n_folds: # noqa: PLR2004 + msg = f"need n_folds >= 2 and n_blocks >= n_folds, got n_folds={n_folds}, n_blocks={n_blocks}" + raise ValueError(msg) + block = np.minimum((np.arange(n) * n_blocks) // max(n, 1), n_blocks - 1) + return (block % n_folds).astype(int) diff --git a/benchmarking/baselines/power_model/matching.py b/benchmarking/baselines/power_model/matching.py new file mode 100644 index 00000000..b98e99d0 --- /dev/null +++ b/benchmarking/baselines/power_model/matching.py @@ -0,0 +1,209 @@ +"""Coarsened exact matching (CEM) for the power-model bias-cancellation correction (Issue 8). + +The bias-cancellation correction trains/predicts in two symmetric directions and relies on a *common +per-bin multiplicative shrinkage* cancelling between them — which only holds if the baseline and +upgraded periods share a covariate distribution within each reporting bin. This module makes that +matching explicit and model-free: bin the matching variables into cells, keep only cells present on +*both* sides (the common-support guard — the exact failure that sank the R-learner in prepost, F1), +and seeded-subsample the larger side down to the smaller within every retained cell so the two sides +carry equal weight per cell. + +Deliberately pure: it takes a matching frame + boolean side masks and returns matched index positions +plus a balance/coverage diagnostic, with no dependency on the method. The matching axis (ERA5) is kept +separate from the reporting/binning axis (test-turbine ws/TI); this utility is generic over whatever +``bin_edges`` it is handed. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass + +import numpy as np +import pandas as pd + +logger = logging.getLogger(__name__) + +# Hard floor on matched rows per side; mirrors PowerModelMethod._MIN_BASELINE_ROWS so decimated +# matching fails loudly rather than silently producing a noisy estimate. +_MIN_MATCHED_ROWS = 10 +# Below this retained fraction on either side, warn (matching is throwing most of the data away). +_WARN_RETAINED_FRACTION = 0.1 + + +@dataclass +class MatchResult: + """Matched positions per side plus the CEM balance/coverage diagnostic. + + :param baseline_positions: sorted integer positions (into the analysis index) of the matched + baseline rows; equal count to ``upgraded_positions`` within every retained cell + :param upgraded_positions: sorted integer positions of the matched upgraded rows + :param per_cell: one row per cell with the per-var bin code(s), the before counts (``n_baseline`` / + ``n_upgraded``) and the after count per side (``n_matched``, 0 for dropped one-sided cells) + :param n_baseline_in: baseline rows entering matching (finite in every matching var) + :param n_upgraded_in: upgraded rows entering matching + """ + + baseline_positions: np.ndarray + upgraded_positions: np.ndarray + per_cell: pd.DataFrame + n_baseline_in: int + n_upgraded_in: int + + @property + def n_matched_per_side(self) -> int: + """Matched rows per side (the effective sample size the two directions are estimated on).""" + return len(self.baseline_positions) + + @property + def retained_fraction_baseline(self) -> float: + """Fraction of the entering baseline rows kept after matching.""" + return self.n_matched_per_side / self.n_baseline_in if self.n_baseline_in else float("nan") + + @property + def retained_fraction_upgraded(self) -> float: + """Fraction of the entering upgraded rows kept after matching.""" + return self.n_matched_per_side / self.n_upgraded_in if self.n_upgraded_in else float("nan") + + @property + def n_cells_two_sided(self) -> int: + """Cells present on both sides (retained).""" + return int(((self.per_cell["n_baseline"] > 0) & (self.per_cell["n_upgraded"] > 0)).sum()) + + @property + def n_cells_one_sided(self) -> int: + """Cells present on only one side (dropped by the common-support guard).""" + return int(((self.per_cell["n_baseline"] == 0) ^ (self.per_cell["n_upgraded"] == 0)).sum()) + + +def coarsened_exact_match( + matching_frame: pd.DataFrame, + *, + baseline_sel: np.ndarray, + upgraded_sel: np.ndarray, + bin_edges: dict[str, list[float]], + seed: int, + min_matched_rows: int = _MIN_MATCHED_ROWS, + warn_retained_fraction: float = _WARN_RETAINED_FRACTION, +) -> MatchResult: + """Match baseline and upgraded rows on coarsened cells of the matching variables. + + :param matching_frame: the matching columns on the analysis index (row *position* is the identity + the returned positions refer to); must contain every key of ``bin_edges`` + :param baseline_sel: boolean mask over the frame's rows selecting normally-operating baseline rows + :param upgraded_sel: boolean mask over the frame's rows selecting normally-operating upgraded rows + :param bin_edges: ``{var: edges}`` per matching variable; a row outside a var's edges (``pd.cut`` + returns NaN) is unmatchable and dropped, just like a non-finite value + :param seed: seed for the per-cell subsample of the larger side + :param min_matched_rows: hard floor; raise if fewer rows per side survive matching + :param warn_retained_fraction: warn if either side retains less than this fraction + """ + variables = list(bin_edges) + missing = [v for v in variables if v not in matching_frame.columns] + if missing: + msg = f"matching_frame is missing matching columns {missing}; have {list(matching_frame.columns)}" + raise ValueError(msg) + + n_rows = len(matching_frame) + for name, sel in (("baseline_sel", baseline_sel), ("upgraded_sel", upgraded_sel)): + if np.asarray(sel).shape != (n_rows,): + msg = f"{name} shape {np.asarray(sel).shape} does not align to matching_frame's {n_rows} rows" + raise ValueError(msg) + + codes = cell_codes(matching_frame, bin_edges) + valid_cell = np.all(codes >= 0, axis=1) # -1 marks a NaN/out-of-range value in some var + base = np.flatnonzero(np.asarray(baseline_sel, dtype=bool) & valid_cell) + up = np.flatnonzero(np.asarray(upgraded_sel, dtype=bool) & valid_cell) + n_baseline_in = int(base.size) + n_upgraded_in = int(up.size) + + per_cell, matched_base, matched_up = _match_cells(codes, variables=variables, base=base, up=up, seed=seed) + + result = MatchResult( + baseline_positions=np.sort(matched_base), + upgraded_positions=np.sort(matched_up), + per_cell=per_cell, + n_baseline_in=n_baseline_in, + n_upgraded_in=n_upgraded_in, + ) + _check_coverage(result, min_matched_rows=min_matched_rows, warn_retained_fraction=warn_retained_fraction) + return result + + +def cell_codes(matching_frame: pd.DataFrame, bin_edges: dict[str, list[float]]) -> np.ndarray: + """Integer cell code per (row, var); -1 where the value is NaN or outside the var's edges. + + Public because the coarsening is shared: CEM matching here, and the residual calibration + (Issue 13), which estimates mean out-of-fold residuals over the same cells. + """ + columns = [] + for var, edges in bin_edges.items(): + # include_lowest so a value exactly on the lowest edge (e.g. 0° direction, calm wind) lands in + # the first cell rather than becoming NaN and being dropped as if out of range. + cut = pd.cut(matching_frame[var], bins=edges, labels=False, include_lowest=True) # NaN outside / on NaN + columns.append(cut.to_numpy(dtype=float)) + stacked = np.column_stack(columns) + return np.where(np.isfinite(stacked), stacked, -1.0).astype(int) + + +def _match_cells( + codes: np.ndarray, *, variables: list[str], base: np.ndarray, up: np.ndarray, seed: int +) -> tuple[pd.DataFrame, np.ndarray, np.ndarray]: + """Group the two sides by cell, drop one-sided cells, subsample the larger side per two-sided cell.""" + rng = np.random.default_rng(seed) + base_by_cell = _positions_by_cell(codes, base) + up_by_cell = _positions_by_cell(codes, up) + + rows: list[dict] = [] + matched_base: list[int] = [] + matched_up: list[int] = [] + for cell in sorted(set(base_by_cell) | set(up_by_cell)): + b = base_by_cell.get(cell, np.empty(0, dtype=int)) + u = up_by_cell.get(cell, np.empty(0, dtype=int)) + k = min(len(b), len(u)) # 0 for a one-sided cell -> nothing kept + if k: + matched_base.extend(_subsample(b, k, rng).tolist()) + matched_up.extend(_subsample(u, k, rng).tolist()) + row = dict(zip(variables, cell, strict=True)) + row |= {"n_baseline": len(b), "n_upgraded": len(u), "n_matched": k} + rows.append(row) + + per_cell = pd.DataFrame(rows, columns=[*variables, "n_baseline", "n_upgraded", "n_matched"]) + return per_cell, np.asarray(matched_base, dtype=int), np.asarray(matched_up, dtype=int) + + +def _positions_by_cell(codes: np.ndarray, positions: np.ndarray) -> dict[tuple[int, ...], np.ndarray]: + """Map each cell key (tuple of per-var codes) to the given positions that fall in it.""" + out: dict[tuple[int, ...], list[int]] = {} + for pos in positions: + out.setdefault(tuple(codes[pos].tolist()), []).append(int(pos)) + return {cell: np.asarray(v, dtype=int) for cell, v in out.items()} + + +def _subsample(positions: np.ndarray, k: int, rng: np.random.Generator) -> np.ndarray: + """Keep all ``k`` positions when the side is already at ``k``, else seeded-choose ``k`` without replacement.""" + if len(positions) == k: + return positions + return rng.choice(positions, size=k, replace=False) + + +def _check_coverage(result: MatchResult, *, min_matched_rows: int, warn_retained_fraction: float) -> None: + """Warn on thin retention; raise below the hard floor so decimated matching fails loudly.""" + if result.n_matched_per_side < min_matched_rows: + msg = ( + f"CEM retained only {result.n_matched_per_side} matched rows per side (< {min_matched_rows}); " + f"the matching cells barely overlap. Coarsen the bins or widen the campaign." + ) + raise ValueError(msg) + for side, frac in ( + ("baseline", result.retained_fraction_baseline), + ("upgraded", result.retained_fraction_upgraded), + ): + if frac < warn_retained_fraction: + logger.warning( + "CEM retained only %.1f%% of the %s rows (%d of the entering set); the matched estimate rests " + "on a small, possibly unrepresentative slice.", + 100 * frac, + side, + result.n_matched_per_side, + ) diff --git a/benchmarking/baselines/power_model/method.py b/benchmarking/baselines/power_model/method.py new file mode 100644 index 00000000..cdb96b94 --- /dev/null +++ b/benchmarking/baselines/power_model/method.py @@ -0,0 +1,1005 @@ +"""``PowerModelMethod``: the simplest-possible ML uplift method behind the harness ``Method`` seam. + +A single supervised **counterfactual power model**: learn the test turbine's normal power as a +function of *reference-only* (upgrade-invariant) features over the baseline period, predict the +counterfactual power over the upgraded period, and take the energy ratio +``uplift = sum(actual) / sum(counterfactual) - 1`` over the upgraded rows. No propensity, no +cross-fitting — fit on the baseline, predict on the disjoint upgraded window, so there is no +in-sample leakage to correct for. + +The feature set is curated to the *causes* of test-turbine power — weather (all raw ERA5) and +wakes (each reference's active power + availability). Expressing expected power *through the +references* forms the test-vs-reference contrast that cancels common-mode seasonal/long-term +drift (the lever the R-learner lacks; findings F1). + +It is v0-independent — ERA5 is supplied as a plain hourly DataFrame (the driver fetches it), so +this package imports nothing from ``wind_up``. +""" + +from __future__ import annotations + +import logging +import tempfile +from dataclasses import dataclass, field +from pathlib import Path +from typing import TYPE_CHECKING, Any + +import numpy as np +import pandas as pd + +from benchmarking.baselines.era5_sync import sync_era5 +from benchmarking.baselines.filtering import NormalOperationFilter +from benchmarking.baselines.power_model import diagnostics as diag +from benchmarking.baselines.power_model.conditional import impute_uncovered_bins, relevel_conditional +from benchmarking.baselines.power_model.features import ( + build_reference_features, + check_reference_only, + era5_feature_frame, + extract_outcome, + reference_mean_wind_speed, + test_condition_signals, +) +from benchmarking.baselines.power_model.fitting import time_block_folds +from benchmarking.baselines.power_model.matching import coarsened_exact_match +from benchmarking.baselines.rlearner.nuisance import make_outcome_model +from benchmarking.diagnostics import DiagnosticContext, stages, write_common_diagnostics, write_run_config +from benchmarking.harness.conditions import ( + CONDITION_BINS, + CONDITIONS, + condition_bins, + energy_ratio_by_bin, + validate_conditions, +) +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.harness.toggle import is_toggle, resolve_toggle, toggle_upgrade_start + +if TYPE_CHECKING: + from benchmarking.synthetic import ColumnSchema + +logger = logging.getLogger(__name__) + +_MIN_BASELINE_ROWS = 10 +_MIN_HOLDOUT_ROWS = 20 # below this, report the in-sample fit (no point splitting off a tiny valid set) +# Time-blocked split shape for the baseline holdout diagnostic (``_holdout_fit`` takes fold 0 as the +# held-out slice): 25 contiguous blocks round-robin over 5 folds, so each fold is ~20% of the rows +# spread across the whole window (seasonally balanced) while staying contiguous at the block scale. +_N_FOLDS = 5 +_N_BLOCKS = 25 +# Adaptive time-decay half-life = this multiple of the campaign's own duration (Issue 15 / F20): a +# scale-free "trust pre-campaign data within ~k campaign-durations" rule that gives a short half-life +# for a short campaign (drift protection) and a long one for a long campaign (use the plentiful recent +# data), reproducing F16's regime map with one mechanism-anchored constant. k=2 -> 1mo~60d, 12mo~730d. +_TIME_DECAY_CAMPAIGN_MULTIPLE = 2.0 +_MIN_TIME_DECAY_DURATION_DAYS = 1.0 # floor the campaign duration so the half-life can't degenerate to 0 + +# The removal-ablation verdict (findings F13): raw Open-Meteo columns the curated feature set does +# better without — redundant thermodynamic derivatives of temperature/humidity and the precipitation +# trio. HoT drivers pass this (with availability_feature=False) as the accepted default; kept here, +# not baked into ``era5_exclude``'s dataclass default, so a non-Open-Meteo ERA5 frame is not broken +# by exclusions it never had. +CURATED_ERA5_EXCLUDE: tuple[str, ...] = ( + "apparent_temperature", + "dew_point_2m", + "precipitation", + "rain", + "snowfall", +) + +# The Issue 12 capacity verdict (findings F14): loosening min_child_samples 200 -> 50 materially +# improves prepost spread/score (placebo ALL Δscore -0.62 pp) at neutral overall P50; 20 overshoots +# into overfit. A power_model-specific tuning — the design-note common params in +# ``make_outcome_model`` (shared with the R-learner) are unchanged; drivers pass this instead. +TUNED_MODEL_PARAMS: dict[str, Any] = {"min_child_samples": 50} + +# Per-reporting-bin matched-count floor for the two-direction conditional combine (Issue 14). Below this +# many matched rows *per side* in a ws/TI reporting bin, the combine overshoots — the F7/F9 sparse-extreme +# tail (e.g. TI (0.45,0.50] swinging -77 to +93 pp between replicates) — so the bin is marked uncovered and +# filled by the physics-informed imputer instead of trusting its noisy shape. Compared against the raw +# per-side matched count today; if the per-bin balance reweighting (A5) is adopted it becomes the Kish +# ESS. The value is chosen on placebo/benchmark evidence across many bins (findings F17), not tuned to the +# one known bad TI bin. +_MIN_BIN_MATCHED_COUNT = 50 + +# F6 matching set + bin widths for the bias-cancellation correction, verified on real HoT by the CEM +# coverage sweep (docs/v1/findings.md F6). wind_speed_100m (dominant) is binned finest, wind_gusts_10m +# coarser, and wind_direction_100m in 20° sectors (a reanalysis direction — finer is finer than the +# signal). Fixed sectors, no wraparound (adjacent sectors are just separate cells, per the CEM utility). +_DEFAULT_MATCHING_VARS: tuple[str, ...] = ("wind_speed_100m", "wind_gusts_10m", "wind_direction_100m") +# This method can report every condition axis: its ws/TI come from the test turbine's own measurements +# (post-treatment, accepted per design-note §3) and its power axis is labelled by the counterfactual +# prediction (F23). The default reports all three, preserving the behaviour of every existing caller. +_SUPPORTED_CONDITIONS: tuple[str, ...] = CONDITIONS +_DEFAULT_MATCHING_BIN_EDGES: dict[str, list[float]] = { + "wind_speed_100m": [float(x) for x in np.arange(0.0, 34.0, 2.0)], # 0,2,…,32 + "wind_gusts_10m": [float(x) for x in np.arange(0.0, 48.0, 3.0)], # 0,3,…,45 + "wind_direction_100m": [float(x) for x in np.arange(0.0, 380.0, 20.0)], # 0,20,…,360 +} + + +def _ratio(actual: np.ndarray, counterfactual: np.ndarray) -> float: + """Energy-ratio ``Σactual / Σcounterfactual - 1`` over finite pairs; NaN if the denominator is 0.""" + finite = np.isfinite(actual) & np.isfinite(counterfactual) + denom = float(counterfactual[finite].sum()) + return float(actual[finite].sum()) / denom - 1.0 if denom != 0 else float("nan") + + +def _combine_uplift(r_fwd: np.ndarray, r_rev: np.ndarray) -> np.ndarray: + """Two-direction uplift ``sqrt((1+r_fwd)/(1+r_rev)) - 1`` (shrinkage cancels); NaN if either 1+r <= 0. + + Under a common per-bin multiplicative shrinkage the forward/reverse ratios are ``(1+u)/s`` and + ``1/(s(1+u))``, so their geometric contrast recovers ``u`` with ``s`` cancelled (design/F5). + """ + a = 1.0 + np.asarray(r_fwd, dtype=float) + b = 1.0 + np.asarray(r_rev, dtype=float) + valid = (a > 0) & (b > 0) + frac = np.divide(a, b, out=np.full(np.broadcast(a, b).shape, np.nan), where=valid) + return np.sqrt(frac, out=np.full_like(frac, np.nan), where=np.isfinite(frac)) - 1.0 + + +def _implied_shrinkage(r_fwd: np.ndarray, r_rev: np.ndarray) -> np.ndarray: + """Implied shrinkage ``s = 1/sqrt((1+r_fwd)(1+r_rev))`` — the bias the correction cancelled (diagnostic).""" + a = 1.0 + np.asarray(r_fwd, dtype=float) + b = 1.0 + np.asarray(r_rev, dtype=float) + valid = (a > 0) & (b > 0) + prod = np.multiply(a, b, out=np.full(np.broadcast(a, b).shape, np.nan), where=valid) + root = np.sqrt(prod, out=np.full_like(prod, np.nan), where=np.isfinite(prod)) + return np.divide(1.0, root, out=np.full_like(root, np.nan), where=np.isfinite(root) & (root != 0)) + + +def _condition_frame( + name: str, + *, + fwd_cond: np.ndarray, + rev_cond: np.ndarray, + full_cond: np.ndarray, + y_fwd: np.ndarray, + pred_fwd: np.ndarray, + y_rev: np.ndarray, + pred_rev: np.ndarray, + actual_full: np.ndarray, + bins: list[float], + one_plus_overall: float, +) -> pd.DataFrame: + """One condition axis' per-bin two-direction uplift, imputed where uncovered and re-leveled to the headline. + + The three ``*_cond`` arrays label each side's rows by the reporting axis: for ws/TI they are the same + signal read off every side; for power they are each side's counterfactual prediction (forward = + ``pred_up``, reverse = ``pred_base``, full = the full-fit counterfactual). Labelling every power side by + its prediction — never by the actual outcome that also sits in that side's ratio numerator — is what + keeps the reverse ratio free of a regression-to-the-mean tilt (see the power-frame note in + ``_conditional_by_bin``). The forward/reverse energy ratios give the shrinkage-free per-bin + shape; uncovered bins are imputed (``impute_uncovered_bins``) and the whole thing is re-leveled onto the + headline by full-upgraded energy so measured + imputed aggregate to ``one_plus_overall`` (F8/F14). + """ + fwd = energy_ratio_by_bin(fwd_cond, y_fwd, pred_fwd, bins=bins) + rev = energy_ratio_by_bin(rev_cond, y_rev, pred_rev, bins=bins) + # full-upgraded actual energy per bin (counterfactual arg unused — only sum_actual is taken) + full = energy_ratio_by_bin(full_cond, actual_full, actual_full, bins=bins) + merged = fwd.merge(rev, on="condition_bin", suffixes=("_fwd", "_rev")).merge( + full[["condition_bin", "sum_actual"]].rename(columns={"sum_actual": "sum_actual_full"}), + on="condition_bin", + ) + # impute_uncovered_bins needs ascending bin order (low ws/TI/power first); the merges above do not + # guarantee it, so pin the canonical bin order before imputing / re-leveling. + order = pd.cut([], bins=bins).categories.astype(str) + merged["condition_bin"] = pd.Categorical(merged["condition_bin"], categories=order, ordered=True) + merged = merged.sort_values("condition_bin").reset_index(drop=True) + + r_fwd = merged["p50_uplift_fwd"].to_numpy() # per-bin energy ratio IS the per-bin r + r_rev = merged["p50_uplift_rev"].to_numpy() + shape = 1.0 + _combine_uplift(r_fwd, r_rev) # shrinkage-free shape (NaN if degenerate) + sum_actual_full = merged["sum_actual_full"].to_numpy() + # A bin is covered only if its shape is finite AND both directions have enough matched rows; + # below the floor the two-direction combine overshoots, so the bin is imputed instead (F7/F9). + per_side = np.minimum(merged["n_records_fwd"].to_numpy(), merged["n_records_rev"].to_numpy()) + measured = np.isfinite(shape) & (per_side >= _MIN_BIN_MATCHED_COUNT) + imputed_shape = impute_uncovered_bins(shape, condition=name, measured=measured, one_plus_overall=one_plus_overall) + releveled = relevel_conditional( + sum_actual_full, imputed_shape, measured=measured, one_plus_overall=one_plus_overall + ) + return pd.DataFrame( + { + "condition": name, + "condition_bin": merged["condition_bin"].astype(str), + "n_records_fwd": merged["n_records_fwd"], + "n_records_rev": merged["n_records_rev"], + "sum_actual": sum_actual_full, # full-upgraded energy per bin (the MWh re-level weight) + "r_fwd": r_fwd, + "r_rev": r_rev, + "implied_shrinkage": _implied_shrinkage(r_fwd, r_rev), + "u_b": shape - 1.0, # measured two-direction shape (NaN where uncovered) + "covered": measured, # per-run diagnostic: measured vs imputed + "p50_uplift": releveled - 1.0, # measured-or-imputed, re-leveled to the headline + } + ) + + +def _infer_timebase(index: pd.DatetimeIndex) -> pd.Timedelta: + """Infer the analysis timebase as the median spacing of the sorted unique timestamps.""" + unique = pd.DatetimeIndex(pd.unique(index)).sort_values() + if len(unique) < 2: # noqa: PLR2004 + return pd.Timedelta(minutes=10) + return pd.Timedelta(np.median(np.diff(unique.to_numpy()))) + + +def _clip_predictions(pred: np.ndarray, *, y_train: np.ndarray, rated_power_kw: float) -> np.ndarray: + """Clip boosted predictions to the physically plausible range of the fitted-on outcome. + + Tree boosting sums trees, so a prediction can drift slightly past the training ``y`` range; the + clip binds only at those extremes. ``lower = min(0, min(y_train))`` floors at 0 for non-negative + training data but never pulls a genuinely-negative observation up; ``upper`` is the rated ceiling + but never below the largest observed training outcome. ``y_train`` is the outcome of the rows the + predicting model was fitted on; ``rated_power_kw`` is that turbine's rating for those rows. + """ + lower = min(0.0, float(np.min(y_train))) + upper = max(float(rated_power_kw), float(np.max(y_train))) + return np.clip(pred, lower, upper) + + +@dataclass +class PowerModelMethod: + """Pluggable counterfactual power-model uplift estimator (prepost and toggle). + + :param columns: **required** source-native column schema — the single source of the method's + column names. It reads the ``active_power`` (the outcome ``Y`` and the reference active-power + feature), ``availability`` (drives the test-turbine downtime filter and is itself a reference + "is it waking" feature), ``wind_speed`` (its reference mean feeds the ERA5 lag sync and the + stuck-filter calm exemption) and ``wind_speed_sd`` (the turbulence-intensity signal) roles; + the remaining roles feed only the shared diagnostics. + :param baseline_rated_power_kw: **required** rated power of the turbine over the data the model is + fitted on — today every fit is on baseline rows, so this is the baseline rating. It caps the + clipped counterfactual predictions. (A future cross-predict direction that trains on upgraded + data would need the upgraded rating; that is out of scope here.) + :param era5_hourly_df: optional raw hourly ERA5 (Open-Meteo columns); added as features when given + :param name: method name shown in the leaderboard + :param out_dir: where per-run folders are written; a temp dir when ``None`` + :param save_plots: also write the diagnostic plots + :param seed: seed for the baseline holdout split and the LightGBM ``random_state`` (a caller-supplied + ``random_state`` in ``model_params`` still wins) + :param model_params: LightGBM overrides passed to the outcome model; merged **over** the tuned + default ``TUNED_MODEL_PARAMS`` (``min_child_samples=50``, F14), so a bare method already carries + the accepted capacity and any key here wins + :param timebase: analysis timebase; inferred from the data when ``None`` + :param conditions: which condition axes to report a per-bin conditional uplift over — any of + ``"ws"`` / ``"ti"`` / ``"power"`` (default: all three). The overall P50 is a single + baseline→upgraded fit regardless; when ``conditions`` is non-empty an extra two-direction, + ERA5-weather-matched (CEM) cross-prediction runs **last** to estimate the conditional shape + (its common per-bin shrinkage cancels), then re-levels onto the headline. That step + **requires ERA5** (the matching axis is the ERA5 columns). Pass ``()`` to skip that + cross-prediction — the expensive part — and return only the overall P50. + :param matching_vars: ERA5 columns matched on for the conditional step (default: the F6 set) + :param matching_bin_edges: per-variable CEM bin edges; the F6 defaults are used when ``None`` + :param reference_stat_cols: *extra* per-reference value columns to carry as features beyond the + active-power minimum, which is always carried via the ``columns`` schema's ``active_power_min`` + role (Issue 11 / F12 — the max/SD companions stay opt-in here; a repeat of the min is deduped) + :param era5_exclude: raw ERA5 columns to drop from the model features (removal-ablation knob; + a dropped direction column also loses its sin/cos companions). Columns used as + ``matching_vars`` cannot be excluded while any ``conditions`` are requested. Defaults to the + accepted ``CURATED_ERA5_EXCLUDE`` set (F13); the **untouched default** is drop-if-present (so a + non-Open-Meteo ERA5 frame lacking those columns is not broken), while an **explicitly-set** + value keeps the strict raise-on-unknown-column typo guard. + :param availability_feature: when ``False`` (the accepted default, F13), drop the per-reference + availability *feature*; the ``availability`` role itself stays required for the downtime filter + :param adaptive_time_decay: when ``True`` (**default**, the Issue 15 self-configuring behaviour) + the headline fit's time-decay half-life is set automatically to + ``_TIME_DECAY_CAMPAIGN_MULTIPLE * campaign_duration_days`` — a short half-life for a short + campaign (down-weight the stale pre-campaign era that dominates a sliver campaign) and a long + one for a long campaign (use the plentiful recent data). This subsumes the fixed default: + the best half-life is regime-dependent (F16 — short helps 1-3-month campaigns in both modes, + long is safe at 12 months), and a campaign-proportional half-life gets both ends right with no + manual tuning. When ``True``, ``time_decay_half_life_days`` must be left ``None``. + :param time_decay_half_life_days: **expert override** (used only when ``adaptive_time_decay=False``): + a *fixed* half-life for the campaign-proximity training weights + ``0.5 ** (days_outside_campaign / half_life)``. Rows inside the campaign interval (for toggle, + the interleaved on and off rows) weigh 1; rows outside decay with their distance to it — so + distant history informs the fit without dominating it (Issue 13's recency weighting; the + alternative to the rejected F11 drift *feature*). ``None`` disables the weighting entirely. + The default self-configuring behaviour is ``adaptive_time_decay=True`` (this left ``None``). + + The toggle headline is always the counterfactual energy ratio ``Σactual/Σprediction - 1``. + """ + + columns: ColumnSchema + baseline_rated_power_kw: float + era5_hourly_df: pd.DataFrame | None = None + name: str = "power_model" + out_dir: Path | None = None + save_plots: bool = False + seed: int = 0 + model_params: dict[str, Any] = field(default_factory=dict) + timebase: pd.Timedelta | None = None + conditions: tuple[str, ...] = _SUPPORTED_CONDITIONS + matching_vars: tuple[str, ...] = _DEFAULT_MATCHING_VARS + matching_bin_edges: dict[str, list[float]] | None = None + reference_stat_cols: tuple[str, ...] = () + era5_exclude: tuple[str, ...] = CURATED_ERA5_EXCLUDE + availability_feature: bool = False + adaptive_time_decay: bool = True + time_decay_half_life_days: float | None = None + + def __post_init__(self) -> None: + """Validate ``columns`` names every role this method reads, and the requested ``conditions``.""" + self.columns.require_roles(("active_power", "active_power_min", "availability", "wind_speed", "wind_speed_sd")) + validate_conditions(self.conditions, supported=_SUPPORTED_CONDITIONS, method_name=self.name) + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Estimate the test turbine's P50 uplift for one campaign and write diagnostics.""" + self._validate_model_config() + scada = mi.scada_df + if self.columns.availability not in scada.columns: + msg = ( + f"the availability column {self.columns.availability!r} (columns.availability) is not in " + f"scada_df; the downtime filter is required for the power model and cannot be skipped." + ) + raise ValueError(msg) + index = pd.DatetimeIndex(pd.unique(scada.index)).sort_values() + timebase = self.timebase if self.timebase is not None else _infer_timebase(scada.index) + n_refs = scada[mi.turbine_col].nunique() - 1 + + y = extract_outcome( + scada, test_wtg=mi.test_wtg, turbine_col=mi.turbine_col, active_power_col=self.columns.active_power + ) + # The per-reference active-power minimum (Issue 11 / F12) is a standard feature carried by the + # schema, so it is always present without per-driver ``reference_stat_cols`` config; extra + # stat columns still append after it (deduped, so a caller repeating the min is harmless). + extra_cols = tuple(dict.fromkeys(c for c in (self.columns.active_power_min, *self.reference_stat_cols) if c)) + features = build_reference_features( + scada, + test_wtg=mi.test_wtg, + turbine_col=mi.turbine_col, + active_power_col=self.columns.active_power, + availability_col=self.columns.availability, + extra_cols=extra_cols, + include_availability=self.availability_feature, + ) + features, era5 = self._add_era5(scada, features, mi=mi, index=index, timebase=timebase) + check_reference_only(features.columns.tolist(), test_wtg=mi.test_wtg) + + toggle_rows = resolve_toggle(mi.upgrade_timing, index) + t = toggle_rows.upgraded + selected = self._select_rows(scada, mi=mi, index=index, y=y, timebase=timebase) + # The counterfactual is fitted on the lenient training baseline (pre-campaign U off-blocks); + # more upgrade-invariant data helps. The conditional matching step below instead uses the + # strict campaign_baseline (off-blocks only), so it never matches pre-campaign against on rows. + baseline_sel = selected & toggle_rows.training_baseline + upgraded_sel = selected & t + if int(baseline_sel.sum()) < _MIN_BASELINE_ROWS: + msg = f"too few normally-operating baseline rows ({int(baseline_sel.sum())}) to fit the power model." + raise ValueError(msg) + if not upgraded_sel.any(): + msg = "no normally-operating upgraded rows to estimate uplift over." + raise ValueError(msg) + + y_arr = y.to_numpy(dtype=float) + weights = self._time_decay_weights( + index, campaign_start=toggle_upgrade_start(mi.upgrade_timing, index), campaign_end=index.max() + ) + fit = self._fit_predict( + features, + y=y_arr, + baseline_sel=baseline_sel, + upgraded_sel=upgraded_sel, + weights=weights, + ) + sum_actual = float(fit["y_upgraded"].sum()) + sum_counter = float(fit["pred_upgraded"].sum()) + uplift = sum_actual / sum_counter - 1.0 if np.isfinite(sum_counter) and sum_counter != 0 else float("nan") + + # ws/TI row-aligned to each segment's residuals, for the overall shrinkage-check diagnostics (cheap, + # plots only). Computed here so the run folder's step-5 residual plots are drawn whether or not the + # optional conditional step runs. + conditions = test_condition_signals( + scada, + test_wtg=mi.test_wtg, + turbine_col=mi.turbine_col, + wind_speed_col=self.columns.wind_speed, + wind_speed_sd_col=self.columns.wind_speed_sd, + ) + cond_upgraded = conditions.iloc[upgraded_sel].reset_index(drop=True) + cond_baseline_valid = conditions.iloc[fit["baseline_valid_pos"]].reset_index(drop=True) + + run_dir = self._run_dir(mi, index) + self._write( + mi, + run_dir=run_dir, + index=index, + timebase=timebase, + t=t, + selected=selected, + upgraded_sel=upgraded_sel, + y=y_arr, + features=features, + fit=fit, + uplift=uplift, + sum_actual=sum_actual, + sum_counter=sum_counter, + n_refs=n_refs, + era5=era5, + cond_upgraded=cond_upgraded, + cond_baseline_valid=cond_baseline_valid, + ) + + # The conditional uplift distribution is the optional, expensive last step: nothing above depends on + # it (eventually AEP extrapolation will). Skipped when no conditions are requested. + by_condition: pd.DataFrame | None = None + if self.conditions: + # The conditional two-direction step matches baseline against upgraded rows and relies on + # them sharing a distribution *and era* — for a toggle whose headline fit also trains on the + # pre-campaign baseline, only the interleaved campaign off rows qualify (the strict + # ``campaign_baseline``): matching pre-campaign rows against campaign on rows would read + # reference/era drift as per-bin uplift. The extra pre-campaign rows serve only the headline + # fit's training data. (F26 re-confirmed this post-floor: unifying onto ``training_baseline`` + # blew up tail-bin |bias|/spread — the added rows promote drift-contaminated bins past the count + # floor from imputed to trusted.) + # No time-decay weights here either (F16, re-confirmed post-floor in F25): the matched contrast + # is already era-insensitive (its common shrinkage cancels), and weighting the direction fits + # only churns the sparse extreme-condition tail bins (and slightly worsens ws spread) without + # helping the populated bins — so the corrections are spent on the overall headline instead. + by_condition = self._estimate_conditional( + scada, + mi=mi, + features=features, + y=y_arr, + baseline_sel=selected & toggle_rows.campaign_baseline, + upgraded_sel=upgraded_sel, + fit=fit, + overall_ratio=uplift, + run_dir=run_dir, + ) + return MethodOutput(p50_overall=uplift, p50_by_condition=by_condition) + + def _estimate_conditional( + self, + scada: pd.DataFrame, + *, + mi: MethodInput, + features: pd.DataFrame, + y: np.ndarray, + baseline_sel: np.ndarray, + upgraded_sel: np.ndarray, + fit: dict[str, Any], + overall_ratio: float, + run_dir: Path, + ) -> pd.DataFrame | None: + """Per-(ws, TI)-bin conditional uplift via a two-direction, ERA5-weather-matched cross-prediction. + + Match the baseline and upgraded periods on ERA5 weather, then fit/predict in both directions and + combine so the common per-bin shrinkage cancels (design/F5); the decomposition is re-leveled onto the + already-computed overall headline (``overall_ratio``, from the single full fit ``fit``) so the per-bin + MWh partitions it (F8). Requires ERA5 — the matching axis is the synced ERA5 columns, which live in + ``features`` (``era5_feature_frame`` passes them through). Returns the ``[condition, condition_bin, + p50_uplift]`` frame (or ``None`` when no wind-speed column), and writes the per-run diagnostics. + """ + if self.era5_hourly_df is None: + msg = ( + "the conditional step requires ERA5 (era5_hourly_df): the matching axis is the ERA5 weather " + "columns. Supply era5_hourly_df, or pass conditions=() for an overall-only estimate." + ) + raise ValueError(msg) + + # Matched two-direction fits, for the per-bin shape only. Matching is required: without it the reverse + # model (train upgraded, predict baseline) would extrapolate out-of-distribution across the prepost + # weather shift. + edges = self.matching_bin_edges if self.matching_bin_edges is not None else self._default_bin_edges() + match = coarsened_exact_match( + features[list(self.matching_vars)], + baseline_sel=baseline_sel, + upgraded_sel=upgraded_sel, + bin_edges=edges, + seed=self.seed, + ) + mb, mu = match.baseline_positions, match.upgraded_positions + # Forward (train matched-baseline, predict matched-upgraded) gives ``1+r_fwd = (1+u)/s``. + pred_up = self._fit_direction(features, y, train=mb, predict=mu) + # Reverse (train matched-upgraded, predict matched-baseline): 1+r_rev = 1/(s(1+u)). Clip reuses + # baseline_rated_power_kw — its upper bound max(rated, max(y_train)) already lifts the ceiling for an + # uprate, so no separate upgraded-rating field is needed. + pred_base = self._fit_direction(features, y, train=mu, predict=mb) + r_fwd = _ratio(y[mu], pred_up) + r_rev = _ratio(y[mb], pred_base) + + per_bin = self._conditional_by_bin( + scada, + mi=mi, + y=y, + mb=mb, + mu=mu, + pred_up=pred_up, + pred_base=pred_base, + upgraded_pos=np.flatnonzero(upgraded_sel), + actual_full=fit["y_upgraded"], + pred_upgraded=fit["pred_upgraded"], + overall_ratio=overall_ratio, + ) + by_condition = per_bin[["condition", "condition_bin", "p50_uplift"]].copy() if per_bin is not None else None + self._write_conditional( + mi, run_dir=run_dir, match=match, r_fwd=r_fwd, r_rev=r_rev, uplift=overall_ratio, per_bin=per_bin + ) + logger.info( + "%s %s conditional uplift: headline=%+.3f%% implied_s=%.3f (r_fwd=%+.3f, r_rev=%+.3f, matched=%d/side)", + self.name, + mi.test_wtg, + 100 * overall_ratio, + float(_implied_shrinkage(np.array([r_fwd]), np.array([r_rev]))[0]), + r_fwd, + r_rev, + match.n_matched_per_side, + ) + return by_condition + + def _fit_direction( + self, features: pd.DataFrame, y: np.ndarray, *, train: np.ndarray, predict: np.ndarray + ) -> np.ndarray: + """Fit the outcome model(s) on ``train`` rows and return clipped predictions for ``predict`` rows. + + No calibration and no time-decay weights here: the two-direction combine cancels the common + per-bin shrinkage by construction, and weighting the matched fits only churns sparse + extreme-condition bins (F16; re-confirmed post-floor in F25) — the corrections are spent on the + overall headline instead. + """ + models = self._fit_models(features.iloc[train], y[train]) + return _clip_predictions( + self._predict_mean(models, features.iloc[predict]), + y_train=y[train], + rated_power_kw=self.baseline_rated_power_kw, + ) + + def _conditional_by_bin( + self, + scada: pd.DataFrame, + *, + mi: MethodInput, + y: np.ndarray, + mb: np.ndarray, + mu: np.ndarray, + pred_up: np.ndarray, + pred_base: np.ndarray, + upgraded_pos: np.ndarray, + actual_full: np.ndarray, + pred_upgraded: np.ndarray, + overall_ratio: float, + ) -> pd.DataFrame | None: + """Per-(ws, TI, power)-bin two-direction uplift, imputed where uncovered and re-leveled onto the headline. + + The measured **shape** ``1+u_b = sqrt((1+r_fwd_b)/(1+r_rev_b))`` comes from the matched + forward/reverse fits; a bin is ``covered`` (trusted) only when that shape is finite (Issue 14 adds + the per-bin matched-count floor to this test). Uncovered bins are filled by + :func:`~benchmarking.baselines.power_model.conditional.impute_uncovered_bins` (ws: bfill then 0 at + rated; ti: the overall uplift) so every bin carries a best estimate rather than a bare NaN. The + re-level **weights** are the *full-upgraded* actual energy per bin (``actual_full`` over + ``upgraded_pos``): imputed bins are pinned and one λ scales the measured bins so measured + imputed + together energy-aggregate to ``1 + overall_ratio`` exactly (F8, corrected for uncovered-bin energy). + The returned frame carries a ``covered`` flag (a per-run diagnostic — the harness seam only sees + ``p50_uplift``). + """ + conditions = test_condition_signals( + scada, + test_wtg=mi.test_wtg, + turbine_col=mi.turbine_col, + wind_speed_col=self.columns.wind_speed, + wind_speed_sd_col=self.columns.wind_speed_sd, + ) + cond_up = conditions.iloc[mu].reset_index(drop=True) # matched-upgraded (forward binning) + cond_base = conditions.iloc[mb].reset_index(drop=True) # matched-baseline (reverse binning) + cond_full = conditions.iloc[upgraded_pos].reset_index(drop=True) # all upgraded (re-level weights) + one_plus_overall = 1.0 + overall_ratio + frames = [ + _condition_frame( + name, + fwd_cond=cond_up[name].to_numpy(), + rev_cond=cond_base[name].to_numpy(), + full_cond=cond_full[name].to_numpy(), + y_fwd=y[mu], + pred_fwd=pred_up, + y_rev=y[mb], + pred_rev=pred_base, + actual_full=actual_full, + bins=CONDITION_BINS[name], + one_plus_overall=one_plus_overall, + ) + for name in ("ws", "ti") + if name in conditions.columns and name in self.conditions + ] + # power: the baseline operating point, and every side is labelled by its *counterfactual + # prediction* — forward (upgraded) rows by ``pred_up``, reverse (baseline) rows by ``pred_base``, + # the full re-level weights by ``pred_upgraded``. The reverse side must NOT be labelled by its + # actual power ``y[mb]``: ``y[mb]`` is also the reverse ratio's numerator, so binning on it selects + # each bin on its own noise and pulls in a regression-to-the-mean tilt (spuriously positive uplift + # at low power, negative at high — flat truth reads as a slope). Labelling both sides by the + # prediction keeps the per-bin shrinkage common so it cancels in the two-direction combine, which is + # the whole premise of ``_combine_uplift``. Edges scale with the baseline rating (§2 of the design), + # so power is not in the fixed ``CONDITION_BINS``. + if "power" in self.conditions: + frames.append( + _condition_frame( + "power", + fwd_cond=pred_up, + rev_cond=pred_base, + full_cond=pred_upgraded, + y_fwd=y[mu], + pred_fwd=pred_up, + y_rev=y[mb], + pred_rev=pred_base, + actual_full=actual_full, + bins=condition_bins("power", rated_power_kw=self.baseline_rated_power_kw), + one_plus_overall=one_plus_overall, + ) + ) + return pd.concat(frames, ignore_index=True) if frames else None + + def _write_conditional( + self, + mi: MethodInput, + *, + run_dir: Path, + match: Any, # noqa: ANN401 - MatchResult + r_fwd: float, + r_rev: float, + uplift: float, + per_bin: pd.DataFrame | None, + ) -> None: + """Write the conditional-step diagnostics (implied shrinkage + CEM balance) and the per-bin plot. + + CSVs go in ``run_dir/conditional/``; the per-bin implied-shrinkage plot (when ``save_plots``) goes in + the step-7 plot folder, keeping the optional conditional outputs separate from the always-on overall + diagnostics in the same run folder. + """ + conditional_dir = run_dir / "conditional" + conditional_dir.mkdir(parents=True, exist_ok=True) + run_name = run_dir.name + ts = pd.Timestamp.utcnow().strftime("%Y%m%d_%H%M%S_%f") + overall = { + "test_wtg": mi.test_wtg, + "mode": "toggle" if is_toggle(mi.upgrade_timing) else "prepost", + "r_fwd": r_fwd, + "r_rev": r_rev, + "uplift_frc": uplift, + "implied_shrinkage": float(_implied_shrinkage(np.array([r_fwd]), np.array([r_rev]))[0]), + } + diag.write_conditional_csvs(conditional_dir, run_name, ts, overall=overall, per_bin=per_bin, match=match) + if self.save_plots and per_bin is not None: + diag.plot_conditional_diagnostics( + run_dir / "plots" / stages.CONDITIONAL_UPLIFT, per_bin, test_wtg=mi.test_wtg + ) + + def _default_bin_edges(self) -> dict[str, list[float]]: + """Per-variable CEM edges for ``matching_vars`` from the F6 defaults; raise on an unknown var.""" + missing = [v for v in self.matching_vars if v not in _DEFAULT_MATCHING_BIN_EDGES] + if missing: + msg = ( + f"no default matching bin edges for {missing}; pass matching_bin_edges explicitly for " + f"non-default matching_vars. Known defaults: {sorted(_DEFAULT_MATCHING_BIN_EDGES)}." + ) + raise ValueError(msg) + return {v: _DEFAULT_MATCHING_BIN_EDGES[v] for v in self.matching_vars} + + def _add_era5( + self, + scada: pd.DataFrame, + features: pd.DataFrame, + *, + mi: MethodInput, + index: pd.DatetimeIndex, + timebase: pd.Timedelta, + ) -> tuple[pd.DataFrame, Any]: + """Sync ERA5 (if supplied) and append its features.""" + if self.era5_hourly_df is None: + return features, None + if self.columns.wind_speed not in scada.columns: + msg = ( + f"wind_speed_col {self.columns.wind_speed!r} is not in scada_df; it provides the reference wind " + f"speed the ERA5 lag sync locks onto. A missing column silently yields an all-NaN reference " + f"wind speed and a meaningless lag, so this is treated as a configuration error." + ) + raise ValueError(msg) + reference_ws = reference_mean_wind_speed( + scada, test_wtg=mi.test_wtg, turbine_col=mi.turbine_col, wind_speed_col=self.columns.wind_speed + ) + result = sync_era5(self.era5_hourly_df, target_index=index, reference_ws=reference_ws, timebase=timebase) + era5_features = era5_feature_frame(result.aligned) + if self.era5_exclude: + missing = sorted(set(self.era5_exclude) - set(era5_features.columns)) + # An explicitly-set era5_exclude keeps the strict typo guard; the promoted class default + # (CURATED_ERA5_EXCLUDE) is drop-if-present, so a non-Open-Meteo ERA5 frame lacking those + # columns is not broken by an exclusion it never had. Identity check = "untouched default". + if missing and self.era5_exclude is not CURATED_ERA5_EXCLUDE: + msg = f"era5_exclude names columns not in the ERA5 features: {missing}" + raise ValueError(msg) + blocked = sorted(set(self.era5_exclude) & set(self.matching_vars)) + if blocked and self.conditions: + msg = ( + f"era5_exclude {blocked} are matching_vars; excluding them as model features would " + f"break the CEM matching cells. Pass conditions=() or re-pick matching_vars first." + ) + raise ValueError(msg) + drop = [ + c + for c in era5_features.columns + if c in self.era5_exclude or any(c == f"{raw}_{t}" for raw in self.era5_exclude for t in ("sin", "cos")) + ] + era5_features = era5_features.drop(columns=drop) + return pd.concat([features, era5_features], axis=1), result + + def _effective_half_life(self, *, campaign_start: pd.Timestamp, campaign_end: pd.Timestamp) -> float | None: + """Return the time-decay half-life (days) for this campaign, or ``None`` when decay is off. + + ``adaptive_time_decay`` (the default) sets it to ``_TIME_DECAY_CAMPAIGN_MULTIPLE`` times the + campaign's own duration (Issue 15 / F20); otherwise the fixed ``time_decay_half_life_days`` + expert override is used verbatim. + """ + if not self.adaptive_time_decay: + return self.time_decay_half_life_days + duration_days = max((campaign_end - campaign_start).total_seconds() / 86400.0, _MIN_TIME_DECAY_DURATION_DAYS) + return _TIME_DECAY_CAMPAIGN_MULTIPLE * duration_days + + def _time_decay_weights( + self, index: pd.DatetimeIndex, *, campaign_start: pd.Timestamp, campaign_end: pd.Timestamp + ) -> np.ndarray | None: + """Exponential campaign-proximity sample weights over the analysis index (``None`` = knob off). + + ``0.5 ** (days_outside_campaign / half_life)`` where the distance is to the campaign + *interval* ``[campaign_start, campaign_end]``: every row inside the campaign (for toggle, + the interleaved on **and** off rows) weighs exactly 1, and rows outside decay with their + distance to it — so distant history still informs the fit without dominating it. With + today's flows only pre-campaign rows exist outside the interval, but the definition is + two-sided on purpose. The half-life is the ``adaptive_time_decay`` campaign-proportional + value by default (:meth:`_effective_half_life`). + """ + half_life = self._effective_half_life(campaign_start=campaign_start, campaign_end=campaign_end) + if half_life is None: + return None + seconds_outside = np.maximum((campaign_start - index).total_seconds(), (index - campaign_end).total_seconds()) + days_outside = np.maximum(seconds_outside / 86400.0, 0.0) + return np.asarray(0.5 ** (days_outside / half_life), dtype=float) + + def _select_rows( + self, scada: pd.DataFrame, *, mi: MethodInput, index: pd.DatetimeIndex, y: pd.Series, timebase: pd.Timedelta + ) -> np.ndarray: + """Boolean over ``index``: normally-operating test rows (cause-not-effect) with finite outcome.""" + test_rows = scada[scada[mi.turbine_col] == mi.test_wtg].sort_index() + keep = NormalOperationFilter( + active_power_col=self.columns.active_power, + wind_speed_col=self.columns.wind_speed, + availability_col=self.columns.availability, + ).keep_mask(test_rows, timebase=timebase) + keep = keep[~keep.index.duplicated()].reindex(index, fill_value=False) + return keep.to_numpy() & np.isfinite(y.to_numpy(dtype=float)) + + def _fit_predict( + self, + features: pd.DataFrame, + *, + y: np.ndarray, + baseline_sel: np.ndarray, + upgraded_sel: np.ndarray, + weights: np.ndarray | None = None, + ) -> dict[str, Any]: + """Fit on baseline, predict the upgraded counterfactual; also a held-out baseline fit metric. + + The final model (for the counterfactual) is fit on **all** baseline rows. A separate model on + a baseline train split predicts a held-out baseline slice, giving an honest fit-quality number + that is not inflated by in-sample optimism. + """ + x_base = features.iloc[baseline_sel] + y_base = y[baseline_sel] + x_up = features.iloc[upgraded_sel] + y_up = y[upgraded_sel] + w_base = weights[baseline_sel] if weights is not None else None + + y_valid, pred_valid, valid_local = self._holdout_fit(x_base, y_base, w_base=w_base) + baseline_valid_pos = np.flatnonzero(baseline_sel)[valid_local] + + models = self._fit_models(x_base, y_base, weights=w_base) + pred_up = _clip_predictions( + self._predict_mean(models, x_up), y_train=y_base, rated_power_kw=self.baseline_rated_power_kw + ) + return { + "model": models[0], + "pred_upgraded": pred_up, + "y_upgraded": y_up, + "y_baseline_valid": y_valid, + "pred_baseline_valid": pred_valid, + "baseline_valid_pos": baseline_valid_pos, # positions over ``index`` of the held-out rows + } + + def _validate_model_config(self) -> None: + """Fail loudly on config combinations that would silently misbehave.""" + if self.time_decay_half_life_days is not None and self.time_decay_half_life_days <= 0: + msg = f"time_decay_half_life_days must be positive, got {self.time_decay_half_life_days}" + raise ValueError(msg) + if self.adaptive_time_decay and self.time_decay_half_life_days is not None: + msg = ( + "adaptive_time_decay sets the half-life from the campaign duration; a fixed " + "time_decay_half_life_days is only used with adaptive_time_decay=False. Set " + "adaptive_time_decay=False to use the fixed override, or leave time_decay_half_life_days=None." + ) + raise ValueError(msg) + + @staticmethod + def _fit_kwargs(weights: np.ndarray | None) -> dict[str, Any]: + """Return the fit kwargs for the time-decay ``weights`` (empty when unweighted).""" + return {} if weights is None else {"sample_weight": weights} + + def _make_model(self, *, seed: int | None = None) -> Any: # noqa: ANN401 + """One unfitted LightGBM outcome model with ``seed`` plumbed in. + + A caller-supplied ``random_state`` in ``model_params`` still wins over ``seed``. + """ + s = self.seed if seed is None else seed + return make_outcome_model(**{"random_state": s, **TUNED_MODEL_PARAMS, **self.model_params}) + + def _fit_models( + self, x_train: pd.DataFrame, y_train: np.ndarray, *, weights: np.ndarray | None = None + ) -> list[Any]: + """Fit the outcome model on one training set. + + ``weights`` (the time-decay sample weights, aligned to the training rows) are passed to the + fit when given. Returns a single-element list so :meth:`_predict_mean` stays uniform. + """ + model = self._make_model(seed=self.seed) + model.fit(x_train, y_train, **self._fit_kwargs(weights)) + return [model] + + @staticmethod + def _predict_mean(models: list[Any], x: pd.DataFrame) -> np.ndarray: + """Mean prediction over the fitted model(s) (unclipped).""" + return np.mean([np.asarray(m.predict(x), dtype=float) for m in models], axis=0) + + def _holdout_fit( + self, x_base: pd.DataFrame, y_base: np.ndarray, *, w_base: np.ndarray | None = None + ) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """Train on a baseline train split, predict a held-out baseline slice (honest fit quality). + + The held-out slice is **time-blocked** (fold 0 of the shared split shape), not shuffled — a + shuffled holdout sits minutes from its training rows, so autocorrelation makes its residuals + optimistic. Also returns the held-out rows' positions **within the baseline block** so the + caller can line the residuals up with their conditions (ws/TI) for the diagnostics. + """ + n = len(y_base) + if n < _MIN_HOLDOUT_ROWS: + models = self._fit_models(x_base, y_base, weights=w_base) + pred = _clip_predictions( + self._predict_mean(models, x_base), + y_train=y_base, + rated_power_kw=self.baseline_rated_power_kw, + ) + return y_base, pred, np.arange(n) + valid = time_block_folds(n, n_folds=_N_FOLDS, n_blocks=_N_BLOCKS) == 0 + models = self._fit_models( + x_base.iloc[~valid], y_base[~valid], weights=w_base[~valid] if w_base is not None else None + ) + pred_valid = _clip_predictions( + self._predict_mean(models, x_base.iloc[valid]), + y_train=y_base[~valid], + rated_power_kw=self.baseline_rated_power_kw, + ) + return y_base[valid], pred_valid, np.flatnonzero(valid) + + def _run_dir(self, mi: MethodInput, index: pd.DatetimeIndex) -> Path: + """Return the per-run output folder ``/power_model___`` (a temp dir when unset). + + Computed once per ``estimate`` and shared by the overall diagnostics and the optional conditional + step so both write into the *same* run folder. + """ + upgrade_start = toggle_upgrade_start(mi.upgrade_timing, index) + run_name = f"power_model_{mi.test_wtg}_{upgrade_start:%Y%m%d}_{index.max():%Y%m%d}" + out_root = Path(self.out_dir) if self.out_dir is not None else Path(tempfile.mkdtemp(prefix="power_model_")) + run_dir = out_root / run_name + run_dir.mkdir(parents=True, exist_ok=True) + return run_dir + + def _write( + self, + mi: MethodInput, + *, + run_dir: Path, + index: pd.DatetimeIndex, + timebase: pd.Timedelta, + t: np.ndarray, + selected: np.ndarray, + upgraded_sel: np.ndarray, + y: np.ndarray, + features: pd.DataFrame, + fit: dict[str, Any], + uplift: float, + sum_actual: float, + sum_counter: float, + n_refs: int, + era5: Any, # noqa: ANN401 + cond_upgraded: pd.DataFrame | None = None, + cond_baseline_valid: pd.DataFrame | None = None, + ) -> None: + """Assemble the diagnostic data and write the CSVs (+ plots), logging the top features.""" + run_name = run_dir.name + ts = pd.Timestamp.utcnow().strftime("%Y%m%d_%H%M%S_%f") + + x_sel = features.iloc[selected] + data = diag.DiagnosticData( + test_wtg=mi.test_wtg, + mode="toggle" if is_toggle(mi.upgrade_timing) else "prepost", + index=index, + treated_all=t, + selected_all=selected, + y_all=y, + timebase=timebase, + upgraded_ts=index[upgraded_sel], + y_upgraded=fit["y_upgraded"], + pred_upgraded=fit["pred_upgraded"], + y_baseline_valid=fit["y_baseline_valid"], + pred_baseline_valid=fit["pred_baseline_valid"], + feature_names=list(features.columns), + feature_values=x_sel, + y_selected=y[selected], + outcome_model=fit["model"], + overall_uplift=uplift, + sum_actual_kw=sum_actual, + sum_counterfactual_kw=sum_counter, + n_refs=n_refs, + era5_lag_rows=era5.best_lag_rows if era5 is not None else None, + era5_corr=era5.best_corr if era5 is not None else None, + era5_sweep=era5.sweep if era5 is not None else None, + cond_upgraded=cond_upgraded, + cond_baseline_valid=cond_baseline_valid, + ) + importance = diag.write_csvs(run_dir, run_name, ts, data) + diag.log_top_features(importance) + logger.info( + "%s %s: uplift=%+.3f%% (sum_actual=%.1f MWh, sum_counterfactual=%.1f MWh, n_up=%d)", + self.name, + mi.test_wtg, + 100 * uplift, + sum_actual * (timebase / pd.Timedelta(hours=1)) / 1000.0, + sum_counter * (timebase / pd.Timedelta(hours=1)) / 1000.0, + len(fit["y_upgraded"]), + ) + if self.save_plots: + diag.save_plots(run_dir / "plots", data, importance) + self._write_shared_diagnostics(mi, run_dir=run_dir, t=t, selected=selected, timebase=timebase, era5=era5) + + def _write_shared_diagnostics( + self, + mi: MethodInput, + *, + run_dir: Path, + t: np.ndarray, + selected: np.ndarray, + timebase: pd.Timedelta, + era5: Any, # noqa: ANN401 + ) -> None: + """Emit the shared cross-method diagnostics (coverage/curves/histograms) and the run config.""" + ctx = DiagnosticContext( + run_dir=run_dir, + test_wtg=mi.test_wtg, + turbine_col=mi.turbine_col, + columns=self.columns, + scada_df=mi.scada_df, + treated_ts=t.astype(bool), + used_ts=np.asarray(selected, dtype=bool), + timebase=timebase, + mode="toggle" if is_toggle(mi.upgrade_timing) else "prepost", + era5_df=era5.aligned if era5 is not None else None, + ) + write_common_diagnostics(ctx) + extra = { + "era5_lag_rows": era5.best_lag_rows if era5 is not None else None, + "era5_corr": era5.best_corr if era5 is not None else None, + } + write_run_config(ctx, method_name=self.name, method_params=self._config_params(), extra=extra) + + def _config_params(self) -> dict[str, Any]: + """Return the power-model configuration recorded in the run-config YAML.""" + return { + "active_power_col": self.columns.active_power, + "availability_col": self.columns.availability, + "baseline_rated_power_kw": self.baseline_rated_power_kw, + "wind_speed_col": self.columns.wind_speed, + "wind_speed_sd_col": self.columns.wind_speed_sd, + "seed": self.seed, + "has_era5": self.era5_hourly_df is not None, + "reference_stat_cols": list(self.reference_stat_cols), + "era5_exclude": list(self.era5_exclude), + "availability_feature": self.availability_feature, + "model_params": {**TUNED_MODEL_PARAMS, **self.model_params}, + "adaptive_time_decay": self.adaptive_time_decay, + "time_decay_half_life_days": self.time_decay_half_life_days, + } diff --git a/benchmarking/baselines/rlearner/__init__.py b/benchmarking/baselines/rlearner/__init__.py new file mode 100644 index 00000000..6de7e025 --- /dev/null +++ b/benchmarking/baselines/rlearner/__init__.py @@ -0,0 +1,11 @@ +"""Cross-fit R-learner uplift method (v1 Issue 5). + +A pluggable, v0-independent treatment-effect estimator behind the harness ``Method`` seam. +See ``docs/v1/issues.md`` (Issue 5); module docstrings cite the ML uplift design note by section. +""" + +from __future__ import annotations + +from benchmarking.baselines.rlearner.method import RLearnerMethod + +__all__ = ["RLearnerMethod"] diff --git a/benchmarking/baselines/rlearner/diagnostics.py b/benchmarking/baselines/rlearner/diagnostics.py new file mode 100644 index 00000000..cd860ee2 --- /dev/null +++ b/benchmarking/baselines/rlearner/diagnostics.py @@ -0,0 +1,394 @@ +"""Per-run diagnostics for the R-learner: CSVs, feature importance, and plots. + +A human reviewer must be able to confirm the right data was received and interpreted, and — +critically — spot feature leakage (a feature that trivially predicts power, e.g. a +post-treatment nacelle wind speed or a series-wired voltage; design note §3). Hence the +feature-importance table and plot are first-class outputs and the top features are logged. + +Everything here is pure reporting; the estimate itself is computed in :mod:`method`. +""" + +from __future__ import annotations + +import logging +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +from matplotlib.colors import Normalize + +from benchmarking.baselines.rlearner.features import QUALIFIER +from benchmarking.diagnostics import stages +from benchmarking.diagnostics.density import density_scatter +from benchmarking.diagnostics.style import apply_grid, save_fig + +if TYPE_CHECKING: + from pathlib import Path + +logger = logging.getLogger(__name__) + +_SEGMENTS = ("all", "baseline", "upgraded") +_MODEL_LABELS = ("outcome", "propensity", "effect") +_TOP_FEATURES_LOGGED = 10 +_MIN_CORR_PAIRS = 2 +# Per-segment colours used across the R-learner diagnostic plots. +_SEGMENT_COLORS = {"all": "C0", "baseline": "C0", "upgraded": "C1"} + + +@dataclass +class DiagnosticData: + """Everything the diagnostics need; assembled by :class:`~.method.RLearnerMethod`.""" + + test_wtg: str + mode: str + index: pd.DatetimeIndex # all unique timestamps + treated_all: np.ndarray # upgrade flag (0/1) over all timestamps + selected_all: np.ndarray # bool: rows used in the fit + y_all: np.ndarray # test power over all timestamps + timebase: pd.Timedelta + tau: np.ndarray # per selected row + m_hat: np.ndarray + e_hat: np.ndarray + mu0: np.ndarray + y_selected: np.ndarray + condition_ws: np.ndarray | None # test turbine's own ws over selected rows (for tau-vs-ws) + condition_ws_label: str | None # the original column name of condition_ws (for axis labels) + feature_names: list[str] + feature_values: pd.DataFrame # the selected feature matrix X (for the feature catalogue) + outcome_model: Any + propensity_model: Any + effect_model: Any + overall_uplift: float + n_refs: int + era5_lag_rows: int | None + era5_corr: float | None + era5_sweep: pd.DataFrame | None + + +def feature_importance_long(data: DiagnosticData) -> pd.DataFrame: + """Long table of LightGBM gain/split importance for the outcome, propensity and effect models.""" + blocks = [] + models = ( + ("outcome", data.outcome_model), + ("propensity", data.propensity_model), + ("effect", data.effect_model), + ) + for label, model in models: + booster = model.booster_ + blocks.append( + pd.DataFrame( + { + "model": label, + "feature": data.feature_names, + "gain": booster.feature_importance(importance_type="gain"), + "split_count": booster.feature_importance(importance_type="split"), + } + ) + ) + return pd.concat(blocks, ignore_index=True).sort_values(["model", "gain"], ascending=[True, False]) + + +def log_top_features(importance: pd.DataFrame) -> None: + """Log the outcome model's top features so a human can spot a leaking (too-good) predictor.""" + top = importance[importance["model"] == "outcome"].head(_TOP_FEATURES_LOGGED) + pairs = ", ".join(f"{r.feature} (gain={r.gain:.0f})" for r in top.itertuples()) + logger.info("R-learner outcome-model top features by gain: %s", pairs) + logger.info("Review the above for leakage: a feature that trivially predicts power is a red flag.") + + +def segment_stats(data: DiagnosticData) -> pd.DataFrame: + """Per-segment (all/baseline/upgraded) counts and energy, for a human data sanity check.""" + timebase_hours = data.timebase / pd.Timedelta(hours=1) + treated = data.treated_all.astype(bool) + masks = {"all": np.ones(len(data.index), dtype=bool), "baseline": ~treated, "upgraded": treated} + rows = [] + for segment in _SEGMENTS: + seg = masks[segment] + seg_sel = seg & data.selected_all + seg_power = data.y_all[seg_sel] + finite = seg_power[np.isfinite(seg_power)] + rows.append( + { + "segment": segment, + "first_timestamp": data.index[seg].min() if seg.any() else pd.NaT, + "last_timestamp": data.index[seg].max() if seg.any() else pd.NaT, + "n_timestamps": int(seg.sum()), + "n_selected": int(seg_sel.sum()), + "selected_fraction": float(seg_sel.sum() / seg.sum()) if seg.any() else np.nan, + "test_mean_power_kw": float(finite.mean()) if len(finite) else np.nan, + "test_mwh": float(finite.sum()) * timebase_hours / 1000.0 if len(finite) else np.nan, + } + ) + return pd.DataFrame(rows) + + +def results_row(data: DiagnosticData) -> pd.DataFrame: + """Single-row headline results: uplift, sizes, nuisance fit quality, ERA5 sync.""" + treated_sel = data.treated_all[data.selected_all].astype(bool) + resid = data.y_selected - data.m_hat + ss_res = float(np.sum(resid**2)) + ss_tot = float(np.sum((data.y_selected - data.y_selected.mean()) ** 2)) + return pd.DataFrame( + [ + { + "test_wtg": data.test_wtg, + "mode": data.mode, + "n_refs": data.n_refs, + "n_timestamps": len(data.index), + "n_selected": int(data.selected_all.sum()), + "n_selected_upgraded": int(treated_sel.sum()), + "n_features": len(data.feature_names), + "uplift_frc": data.overall_uplift, + "outcome_mae": float(np.mean(np.abs(resid))), + "outcome_r2": 1.0 - ss_res / ss_tot if ss_tot else np.nan, + "propensity_mean": float(np.mean(data.e_hat)), + "propensity_std": float(np.std(data.e_hat)), + "era5_lag_rows": data.era5_lag_rows, + "era5_corr": data.era5_corr, + "time_calculated": pd.Timestamp.utcnow(), + } + ] + ) + + +def feature_catalogue(data: DiagnosticData) -> pd.DataFrame: + """One row per ML feature: source tag/turbine, coverage, basic stats, per-model gain, |corr| with power. + + The overview a human scans to decide whether a feature should be dropped (low coverage and/or + no importance) or whether something is missing — with dozens of columns a sortable table is the + practical medium, complemented by :func:`save_plots`'s overview scatter. + """ + importance = feature_importance_long(data) + gain = {model: importance[importance["model"] == model].set_index("feature")["gain"] for model in _MODEL_LABELS} + y = data.y_selected + rows = [] + for feature in data.feature_names: + col = data.feature_values[feature].to_numpy(dtype=float) + finite = np.isfinite(col) + tag, _, turbine = feature.partition(QUALIFIER) + rows.append( + { + "feature": feature, + "source_tag": tag, + "turbine": turbine or "ERA5/derived", + "coverage_pct": float(100.0 * finite.mean()) if len(col) else np.nan, + "mean": float(np.nanmean(col)) if finite.any() else np.nan, + "std": float(np.nanstd(col)) if finite.any() else np.nan, + "min": float(np.nanmin(col)) if finite.any() else np.nan, + "max": float(np.nanmax(col)) if finite.any() else np.nan, + "gain_outcome": float(gain["outcome"].get(feature, 0.0)), + "gain_propensity": float(gain["propensity"].get(feature, 0.0)), + "gain_effect": float(gain["effect"].get(feature, 0.0)), + "abs_corr_with_power": _abs_corr(col, y), + } + ) + return pd.DataFrame(rows).sort_values("gain_outcome", ascending=False, ignore_index=True) + + +def _abs_corr(col: np.ndarray, y: np.ndarray) -> float: + """Absolute Pearson correlation of a feature with the outcome over their finite pairs.""" + pair = np.isfinite(col) & np.isfinite(y) + if pair.sum() < _MIN_CORR_PAIRS or np.std(col[pair]) == 0 or np.std(y[pair]) == 0: + return float("nan") + return float(abs(np.corrcoef(col[pair], y[pair])[0, 1])) + + +def write_csvs(run_dir: Path, run_name: str, ts: str, data: DiagnosticData) -> pd.DataFrame: + """Write the data-stats, results, feature-importance and feature-catalogue CSVs; return importance.""" + segment_stats(data).to_csv(run_dir / f"{run_name}_data_stats_{ts}.csv", index=False) + results_row(data).to_csv(run_dir / f"{run_name}_results_{ts}.csv", index=False) + importance = feature_importance_long(data) + importance.to_csv(run_dir / f"{run_name}_feature_importance_{ts}.csv", index=False) + feature_catalogue(data).to_csv(run_dir / f"{run_name}_feature_catalogue_{ts}.csv", index=False) + return importance + + +def save_plots(plots_dir: Path, data: DiagnosticData, importance: pd.DataFrame) -> None: + """Write the R-learner modelling plots into their analysis-stage subfolders.""" + model_dir = plots_dir / stages.UPLIFT_MODELLING + model_dir.mkdir(parents=True, exist_ok=True) + _plot_importance(model_dir, importance) + _plot_residual_vs_prediction(model_dir, data) + _plot_propensity(model_dir, data) + _plot_predicted_vs_actual(model_dir, data) + _plot_tau(model_dir, data) + + inputs_dir = plots_dir / stages.UPLIFT_INPUTS + inputs_dir.mkdir(parents=True, exist_ok=True) + _plot_feature_overview(inputs_dir, feature_catalogue(data)) + + if data.era5_sweep is not None: + feat_dir = plots_dir / stages.FEATURE_ENG + feat_dir.mkdir(parents=True, exist_ok=True) + _plot_era5_sweep(feat_dir, data) + + +def _plot_feature_overview(plots_dir: Path, catalogue: pd.DataFrame) -> None: + """All features as a horizontal bar of outcome gain, coloured by coverage — the add/remove overview.""" + df = catalogue.sort_values("gain_outcome", ascending=True) + norm = Normalize(vmin=0.0, vmax=100.0) + cmap = plt.get_cmap("viridis") + fig, ax = plt.subplots(figsize=(12, max(6.0, 0.25 * len(df)))) + ax.barh(df["feature"], df["gain_outcome"], color=cmap(norm(df["coverage_pct"].to_numpy()))) + fig.colorbar(plt.cm.ScalarMappable(norm=norm, cmap=cmap), ax=ax, label="coverage [%]") + ax.set_xlabel("outcome-model gain") + ax.set_title("all ML features: importance (bar) and coverage (colour) — short + cold = drop candidate") + apply_grid(ax) + save_fig(fig, plots_dir / "feature_overview.png") + + +def _segment_masks(data: DiagnosticData) -> dict[str, np.ndarray]: + """Boolean masks over the *selected* rows for the all/baseline/upgraded segments.""" + treated_sel = data.treated_all[data.selected_all].astype(bool) + return {"all": np.ones_like(treated_sel, dtype=bool), "baseline": ~treated_sel, "upgraded": treated_sel} + + +def _r2_mae(actual: np.ndarray, predicted: np.ndarray) -> tuple[float, float]: + """Return (R², MAE) over the finite pairs of ``actual`` / ``predicted``.""" + finite = np.isfinite(actual) & np.isfinite(predicted) + actual, predicted = actual[finite], predicted[finite] + if len(actual) == 0: + return float("nan"), float("nan") + resid = actual - predicted + ss_res = float(np.sum(resid**2)) + ss_tot = float(np.sum((actual - actual.mean()) ** 2)) + r2 = 1.0 - ss_res / ss_tot if ss_tot else float("nan") + return r2, float(np.mean(np.abs(resid))) + + +def _plot_importance(plots_dir: Path, importance: pd.DataFrame) -> None: + """Top features by gain for the three models, stacked vertically so long tag names never overlap.""" + panels = ( + ("outcome", "outcome m(x) = E[Y|X] (predicts power)", "C0"), + ("propensity", "propensity e(x) = E[T|X] (predicts upgrade flag)", "C2"), + ("effect", "effect tau(x) (predicts per-record uplift)", "C1"), + ) + fig, axes = plt.subplots(3, 1, figsize=(11, 16)) + for ax, (label, title, color) in zip(axes, panels, strict=True): + top = importance[importance["model"] == label].head(15).iloc[::-1] + ax.barh(top["feature"], top["gain"], color=color) + ax.set_title(title) + ax.set_xlabel("gain") + apply_grid(ax) + fig.suptitle("R-learner feature importance — review for leakage (a feature that trivially predicts power)") + save_fig(fig, plots_dir / "feature_importance.png") + + +def _plot_residual_vs_prediction(plots_dir: Path, data: DiagnosticData) -> None: + """Outcome residual vs prediction, density-coloured, per all/baseline/upgraded segment.""" + masks = _segment_masks(data) + resid = data.y_selected - data.m_hat + fig, axes = plt.subplots(1, 3, figsize=(21, 6), sharex=True, sharey=True) + for i, (ax, segment) in enumerate(zip(axes, _SEGMENTS, strict=True)): + seg = masks[segment] + density_scatter(data.m_hat[seg], resid[seg], ax=ax, s=6, colorbar=(i == len(_SEGMENTS) - 1)) + ax.axhline(0, color="k", linewidth=1) + r2, mae = _r2_mae(data.y_selected[seg], data.m_hat[seg]) + ax.set_title(f"{segment} (R²={r2:.3f}, MAE={mae:.0f} kW, n={int(seg.sum())})") + ax.set_xlabel("predicted power [kW]") + ax.set_ylabel("residual (actual - predicted) [kW]") + apply_grid(ax) + fig.suptitle(f"{data.test_wtg}: outcome residual vs prediction (conditional-bias check)") + save_fig(fig, plots_dir / "residual_vs_prediction.png") + + +def _plot_propensity(plots_dir: Path, data: DiagnosticData) -> None: + """Propensity overlap histogram, y-axis in hours of data (feedback 7).""" + hours_per_row = data.timebase / pd.Timedelta(hours=1) + weights = np.full(len(data.e_hat), hours_per_row) + fig, ax = plt.subplots(figsize=(8, 6)) + ax.hist(data.e_hat, bins=40, range=(0, 1), weights=weights, color="C2") + ax.set_xlabel("propensity e(x) = P(upgraded | X)") + ax.set_ylabel("hours of data") + ax.set_title(f"{data.test_wtg}: propensity overlap (~0.5 for toggle, spread for before/after)") + apply_grid(ax) + save_fig(fig, plots_dir / "propensity_hist.png") + + +def _plot_predicted_vs_actual(plots_dir: Path, data: DiagnosticData) -> None: + """Predicted vs actual power, density-coloured, per all/baseline/upgraded with R²/MAE and a 1:1 line.""" + masks = _segment_masks(data) + lim = [0.0, float(np.nanmax(data.y_selected))] if len(data.y_selected) else [0.0, 1.0] + fig, axes = plt.subplots(1, 3, figsize=(21, 6.5), sharex=True, sharey=True) + for i, (ax, segment) in enumerate(zip(axes, _SEGMENTS, strict=True)): + seg = masks[segment] + density_scatter(data.y_selected[seg], data.m_hat[seg], ax=ax, s=6, colorbar=(i == len(_SEGMENTS) - 1)) + ax.plot(lim, lim, color="red", linewidth=1.2, label="1:1") + r2, mae = _r2_mae(data.y_selected[seg], data.m_hat[seg]) + ax.set_title(f"{segment} (R²={r2:.3f}, MAE={mae:.0f} kW, n={int(seg.sum())})") + ax.set_xlabel("actual power [kW]") + ax.set_ylabel("predicted power [kW]") + ax.legend(loc="upper left") + apply_grid(ax) + fig.suptitle(f"{data.test_wtg}: outcome model predicted vs actual") + save_fig(fig, plots_dir / "predicted_vs_actual.png") + + +def _plot_tau(plots_dir: Path, data: DiagnosticData) -> None: + """Per-record uplift in kW and as % of expected power, with the wind-speed dependence. + + The kW values can be implausibly wide in prepost — that spread is the F1 overlap/extrapolation + symptom (the effect model extrapolating where ``t_res ≈ 0``), not a unit error; the mean is + annotated and the % view normalises by the baseline expected power ``mu0``. + """ + with np.errstate(divide="ignore", invalid="ignore"): + tau_pct = np.where(np.abs(data.mu0) > 0, 100.0 * data.tau / data.mu0, np.nan) + fig, axes = plt.subplots(1, 3, figsize=(20, 6)) + + axes[0].hist(data.tau, bins=40, color="C3") + axes[0].axvline(float(np.nanmean(data.tau)), color="k", linestyle="--", label=f"mean={np.nanmean(data.tau):.1f} kW") + axes[0].set_xlabel("tau(x): absolute uplift [kW]") + axes[0].set_ylabel("count") + axes[0].set_title(f"{data.test_wtg}: per-record uplift [kW]") + axes[0].legend() + apply_grid(axes[0]) + + finite_pct = tau_pct[np.isfinite(tau_pct)] + axes[1].hist(finite_pct, bins=40, range=_robust_range(finite_pct), color="C3") + axes[1].set_xlabel("tau(x) / mu0: uplift [% of expected power]") + axes[1].set_ylabel("count") + axes[1].set_title("per-record uplift [%]") + apply_grid(axes[1]) + + if data.condition_ws is not None: + density_scatter(data.condition_ws, data.tau, ax=axes[2], s=6, colorbar=True) + axes[2].set_xlabel(data.condition_ws_label or "wind speed [m/s]") + axes[2].set_ylabel("tau(x) [kW]") + axes[2].set_title("uplift vs wind speed") + apply_grid(axes[2]) + else: + axes[2].set_visible(False) + save_fig(fig, plots_dir / "tau.png") + + +def _robust_range(values: np.ndarray) -> tuple[float, float] | None: + """Return a 1st-99th percentile range so a few extreme tau% values do not flatten the histogram.""" + if values.size == 0: + return None + lo, hi = np.percentile(values, [1, 99]) + return (float(lo), float(hi)) if hi > lo else None + + +def _plot_era5_sweep(plots_dir: Path, data: DiagnosticData) -> None: + """ERA5 correlation-vs-lag sweep, grid on, with the chosen optimal shift annotated (feedback 2).""" + sweep = data.era5_sweep + if sweep is None: + return + fig, ax = plt.subplots(figsize=(8, 6)) + ax.plot(sweep["shift_rows"], sweep["corr"], marker=".") + if data.era5_lag_rows is not None: + corr_text = f"{data.era5_corr:.3f}" if data.era5_corr is not None else "n/a" + ax.axvline( + data.era5_lag_rows, + color="k", + linestyle="--", + label=f"best shift = {data.era5_lag_rows} rows (corr = {corr_text})", + ) + ax.legend() + ax.set_xlabel("ERA5 shift [rows]") + ax.set_ylabel("wind-speed correlation") + ax.set_title(f"{data.test_wtg}: ERA5-SCADA correlation vs lag") + apply_grid(ax) + save_fig(fig, plots_dir / "era5_sync.png") diff --git a/benchmarking/baselines/rlearner/era5_sync.py b/benchmarking/baselines/rlearner/era5_sync.py new file mode 100644 index 00000000..ef90c0d4 --- /dev/null +++ b/benchmarking/baselines/rlearner/era5_sync.py @@ -0,0 +1,25 @@ +"""ERA5 alignment for the R-learner. + +The sync now lives in :mod:`benchmarking.baselines.era5_sync` because more than one method uses +it; this module re-exports it for backwards compatibility. +""" + +from __future__ import annotations + +from benchmarking.baselines.era5_sync import ( + ERA5_WD, + ERA5_WS, + Era5SyncResult, + find_best_lag, + sync_era5, + upsample_era5_to_timebase, +) + +__all__ = [ + "ERA5_WD", + "ERA5_WS", + "Era5SyncResult", + "find_best_lag", + "sync_era5", + "upsample_era5_to_timebase", +] diff --git a/benchmarking/baselines/rlearner/features.py b/benchmarking/baselines/rlearner/features.py new file mode 100644 index 00000000..618ef600 --- /dev/null +++ b/benchmarking/baselines/rlearner/features.py @@ -0,0 +1,131 @@ +"""Build the R-learner's upgrade-invariant feature matrix from long SCADA. + +The discipline (design note §3): every model feature must be *upgrade-invariant* — derived +from reference turbines (or ERA5 / met-mast / LiDAR), never the test turbine's own signals, +which the upgrade distorts. Within that rule the matrix is **maximal, not curated**: every +source-native column of every reference turbine is used as-is, original tag names intact, so +the model sees all the data holistically and the feature-importance diagnostics name real +tags. + +The only column that must be identified by config is the **test turbine's active power** +(the outcome ``Y``); reference turbines need no per-column configuration. + +Feature columns are named ``"{QUALIFIER}"`` (e.g. ``"wtc_ActPower_mean @ R1"``) +so the original tag name is preserved verbatim. :func:`check_upgrade_invariant` is the +enforcement guard that rejects any test-turbine-qualified column — exercised by the +bias-guard regression test (design note §8). +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd + +from benchmarking.baselines.rlearner.era5_sync import ERA5_WD, ERA5_WS +from benchmarking.synthetic import ToggleSchedule, treated_mask + +# Separator between a source-native tag and the turbine it came from in a feature name. +QUALIFIER = " @ " + + +def _references(scada_df: pd.DataFrame, *, test_wtg: str, turbine_col: str) -> list[str]: + """Sorted reference turbine names (every turbine present except the test turbine).""" + refs = sorted(t for t in scada_df[turbine_col].unique() if t != test_wtg) + if not refs: + msg = ( + f"no reference turbines available for test_wtg {test_wtg!r}: scada_df contains only " + f"{sorted(scada_df[turbine_col].unique())}. The R-learner needs at least one reference turbine." + ) + raise ValueError(msg) + return refs + + +def build_reference_features(scada_df: pd.DataFrame, *, test_wtg: str, turbine_col: str) -> pd.DataFrame: + """Wide upgrade-invariant features: every reference turbine's every value column, NaN-preserving. + + Columns are ``"{QUALIFIER}"`` keeping the original tag name. The test turbine + contributes nothing here (its outcome is extracted separately). NaNs are preserved (no + complete-case dropping) — LightGBM handles them natively. Includes the (currently no-op) + engineered-feature seam. Raises if no reference turbine is present, or (defensively) if any + test-turbine column would leak in. + """ + refs = _references(scada_df, test_wtg=test_wtg, turbine_col=turbine_col) + value_cols = [c for c in scada_df.columns if c != turbine_col] + index = pd.DatetimeIndex(pd.unique(scada_df.index)).sort_values() + + # One pivot over all value columns (MultiIndex columns: (value_col, turbine)) rather than one + # pivot_table per column — much cheaper on wide SCADA frames. Then keep reference turbines only, + # in (value_col, ref) order, and flatten to the " @ " names. + tmp = scada_df.copy() + tmp["_ts"] = scada_df.index + wide = tmp.pivot_table(index="_ts", columns=turbine_col, values=value_cols, aggfunc="first") + keep = [(col, r) for col in value_cols for r in refs if (col, r) in wide.columns] + features = wide.loc[:, keep] + features.columns = [f"{col}{QUALIFIER}{r}" for col, r in keep] + features = features.reindex(index) + features = pd.concat( + [features, engineered_reference_features(scada_df, test_wtg=test_wtg, turbine_col=turbine_col)], axis=1 + ) + features.index.name = index.name + check_upgrade_invariant(features.columns.tolist(), test_wtg=test_wtg) + return features + + +def engineered_reference_features(scada_df: pd.DataFrame, *, test_wtg: str, turbine_col: str) -> pd.DataFrame: # noqa: ARG001 + """Feature-engineering seam — currently a no-op (returns no columns). + + This is the obvious home for *derived* upgrade-invariant features. The prime future + candidate is **north-corrected reference yaw position** (an accurate per-record wind + direction and waking relationship — a key v0 value-add), alongside shear, stability + proxies and air density. Not implemented now: ERA5 wind direction already gives reasonable + directional information, and proper northing would pull in v0's machinery. Returns an empty + frame on the data's unique timestamps so callers can concatenate it unconditionally. + """ + index = pd.DatetimeIndex(pd.unique(scada_df.index)).sort_values() + return pd.DataFrame(index=index) + + +def extract_outcome_and_treatment( + scada_df: pd.DataFrame, + *, + test_wtg: str, + turbine_col: str, + active_power_col: str, + upgrade_timing: pd.Timestamp | ToggleSchedule, +) -> tuple[pd.Series, pd.Series]: + """Return the outcome ``y`` (test turbine power) and upgrade flag ``t`` on the unique index. + + ``y`` is the test turbine's active power (may contain NaN downtime). ``t`` is the integer + upgrade flag from :func:`treated_mask` (1 = upgraded record, 0 = baseline). + """ + index = pd.DatetimeIndex(pd.unique(scada_df.index)).sort_values() + test_rows = scada_df[scada_df[turbine_col] == test_wtg] + y = test_rows[active_power_col].copy() + y.index = pd.DatetimeIndex(test_rows.index) + y = y.reindex(index) + t = pd.Series(np.asarray(treated_mask(index, upgrade_timing)).astype(int), index=index) + return y, t + + +def era5_features(aligned_era5: pd.DataFrame) -> pd.DataFrame: + """Turn aligned ERA5 (ws + wd degrees) into model features: ws passthrough, wd as sin/cos.""" + rad = np.deg2rad(aligned_era5[ERA5_WD].to_numpy(dtype=float)) + return pd.DataFrame( + { + ERA5_WS: aligned_era5[ERA5_WS].to_numpy(dtype=float), + "era5_wd_sin": np.sin(rad), + "era5_wd_cos": np.cos(rad), + }, + index=aligned_era5.index, + ) + + +def check_upgrade_invariant(feature_names: list[str], *, test_wtg: str) -> None: + """Raise if any feature is qualified with the test turbine (violating the §3 rule).""" + offenders = [f for f in feature_names if f.endswith(f"{QUALIFIER}{test_wtg}")] + if offenders: + msg = ( + f"upgrade-invariant rule violated: features derived from the test turbine {test_wtg!r} " + f"are not allowed (the upgrade distorts its signals, design note §3): {offenders}" + ) + raise ValueError(msg) diff --git a/benchmarking/baselines/rlearner/filtering.py b/benchmarking/baselines/rlearner/filtering.py new file mode 100644 index 00000000..c1949eb8 --- /dev/null +++ b/benchmarking/baselines/rlearner/filtering.py @@ -0,0 +1,11 @@ +"""Test-turbine normal-operation filtering for the R-learner. + +The filter now lives in :mod:`benchmarking.baselines.filtering` because more than one method uses +it; this module re-exports it for backwards compatibility. +""" + +from __future__ import annotations + +from benchmarking.baselines.filtering import NormalOperationFilter + +__all__ = ["NormalOperationFilter"] diff --git a/benchmarking/baselines/rlearner/method.py b/benchmarking/baselines/rlearner/method.py new file mode 100644 index 00000000..1f588526 --- /dev/null +++ b/benchmarking/baselines/rlearner/method.py @@ -0,0 +1,339 @@ +"""``RLearnerMethod``: the cross-fit R-learner behind the harness ``Method`` seam. + +Orchestrates the pieces (design note §4): build upgrade-invariant reference features, sync ERA5 +(optional), filter the test turbine to normal operation, cross-fit the R-learner, aggregate to a +single P50 uplift, and write diagnostics. It is v0-independent — ERA5 is supplied as a plain +hourly DataFrame (the driver fetches it), so this package imports nothing from ``wind_up``. + +Aggregation: the overall uplift is ``sum(tau) / sum(mu0)`` over the **upgraded** rows, which +equals the ground-truth definition ``sum(upgraded power)/sum(baseline power) - 1`` because +``mu0`` is the baseline expected power and ``mu0 + tau`` the upgraded. +""" + +from __future__ import annotations + +import logging +import tempfile +from dataclasses import dataclass, field, replace +from pathlib import Path +from typing import TYPE_CHECKING, Any + +import numpy as np +import pandas as pd + +from benchmarking.baselines.rlearner import diagnostics as diag +from benchmarking.baselines.rlearner.era5_sync import sync_era5 +from benchmarking.baselines.rlearner.features import ( + QUALIFIER, + build_reference_features, + era5_features, + extract_outcome_and_treatment, +) +from benchmarking.baselines.rlearner.filtering import NormalOperationFilter +from benchmarking.baselines.rlearner.nuisance import make_effect_model, make_outcome_model, make_propensity_model +from benchmarking.baselines.rlearner.rlearner import cross_fit_rlearner +from benchmarking.diagnostics import DiagnosticContext, write_common_diagnostics, write_run_config +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.synthetic import HOT_COLUMNS, ToggleSchedule + +if TYPE_CHECKING: + from collections.abc import Callable + + from benchmarking.synthetic import ColumnSchema + +logger = logging.getLogger(__name__) + +_MIN_FOLDS = 2 +_MIN_POINTS_FOR_TIMEBASE = 2 + + +def _infer_timebase(index: pd.DatetimeIndex) -> pd.Timedelta: + """Infer the analysis timebase as the median spacing of the sorted unique timestamps.""" + unique = pd.DatetimeIndex(pd.unique(index)).sort_values() + if len(unique) < _MIN_POINTS_FOR_TIMEBASE: + return pd.Timedelta(minutes=10) + return pd.Timedelta(np.median(np.diff(unique.to_numpy()))) + + +def _upgrade_start(upgrade_timing: pd.Timestamp | ToggleSchedule, index: pd.DatetimeIndex) -> pd.Timestamp: + """Return the upgrade-start timestamp (changeover for prepost; toggle origin for toggle).""" + if isinstance(upgrade_timing, ToggleSchedule): + return upgrade_timing.start if upgrade_timing.start is not None else index.min() + return pd.Timestamp(upgrade_timing) + + +def _restrict_to_campaign(mi: MethodInput, *, toggle_campaign_only: bool) -> MethodInput: + """Drop pre-campaign rows for a toggle campaign so the on/off comparison shares a distribution. + + The harness toggle window also carries the pre-campaign baseline, whose distribution differs + from the campaign and reintroduces the temporal confounding toggling exists to avoid. When + ``toggle_campaign_only`` (the default), restrict a toggle input to records at/after the toggle + start (the interleaved on/off blocks), giving a balanced propensity and pure variance + reduction. No-op for prepost and when the flag is off. + """ + timing = mi.upgrade_timing + if not (toggle_campaign_only and isinstance(timing, ToggleSchedule) and timing.start is not None): + return mi + return replace(mi, scada_df=mi.scada_df.loc[mi.scada_df.index >= timing.start]) + + +@dataclass +class RLearnerMethod: + """Pluggable cross-fit R-learner uplift estimator (prepost and toggle). + + :param active_power_col: the test turbine's active-power column (the outcome ``Y``) + :param availability_col: **required** "ready to operate" counter for the downtime filter; rows + below a full period of availability are dropped. Required so downtime filtering can never be + silently skipped (the lack of it was a real oversight). + :param wind_speed_col: the wind-speed tag, used for ERA5 sync (reference mean) and the + stuck-filter low-wind exemption; required if ``era5_hourly_df`` is given + :param era5_hourly_df: optional raw hourly ERA5 (Open-Meteo columns); added as features when given + :param columns: source-native column schema, used only by the shared diagnostics (not estimation) + :param name: method name shown in the leaderboard + :param out_dir: where per-run folders are written; a temp dir when ``None`` + :param save_plots: also write the diagnostic plots + :param n_folds: cross-fitting folds + :param seed: cross-fitting seed + :param model_params: LightGBM overrides passed to every nuisance/effect model + :param timebase: analysis timebase; inferred from the data when ``None`` + :param toggle_campaign_only: for a toggle campaign, fit only on the interleaved on/off blocks + (drop the pre-campaign baseline) so the propensity is balanced and there is no temporal + confounding; no-op for prepost (whose baseline is the pre-campaign data) + """ + + active_power_col: str + availability_col: str + wind_speed_col: str | None = None + era5_hourly_df: pd.DataFrame | None = None + columns: ColumnSchema = HOT_COLUMNS + name: str = "rlearner" + out_dir: Path | None = None + save_plots: bool = False + n_folds: int = 5 + seed: int = 0 + model_params: dict[str, Any] = field(default_factory=dict) + timebase: pd.Timedelta | None = None + toggle_campaign_only: bool = True + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Estimate the test turbine's P50 uplift for one campaign and write diagnostics.""" + mi = _restrict_to_campaign(mi, toggle_campaign_only=self.toggle_campaign_only) + scada = mi.scada_df + if self.availability_col not in scada.columns: + msg = ( + f"availability_col {self.availability_col!r} is not in scada_df; the downtime filter is " + f"required for the R-learner and cannot be skipped." + ) + raise ValueError(msg) + index = pd.DatetimeIndex(pd.unique(scada.index)).sort_values() + timebase = self.timebase if self.timebase is not None else _infer_timebase(scada.index) + n_refs = scada[mi.turbine_col].nunique() - 1 + + y, t = extract_outcome_and_treatment( + scada, + test_wtg=mi.test_wtg, + turbine_col=mi.turbine_col, + active_power_col=self.active_power_col, + upgrade_timing=mi.upgrade_timing, + ) + features = build_reference_features(scada, test_wtg=mi.test_wtg, turbine_col=mi.turbine_col) + reference_ws = self._reference_mean_ws(features, index=index) + own_ws = self._own_ws(scada, mi=mi, index=index) + features, era5 = self._add_era5(features, index=index, reference_ws=reference_ws, timebase=timebase) + + selected = self._select_rows(scada, mi=mi, index=index, y=y, timebase=timebase) + x_sel = features.loc[index[selected]] + y_sel = y.to_numpy(dtype=float)[selected] + t_sel = t.to_numpy(dtype=float)[selected] + n_folds = min(self.n_folds, int(selected.sum())) + if n_folds < _MIN_FOLDS: + msg = f"too few normally-operating rows ({int(selected.sum())}) to cross-fit the R-learner." + raise ValueError(msg) + + fit = cross_fit_rlearner(x_sel, y=y_sel, t=t_sel, n_folds=n_folds, seed=self.seed, **self._factories()) + overall = _aggregate_uplift(tau=fit.tau, mu0=fit.mu0, upgraded=t_sel.astype(bool)) + + self._write( + mi, + index=index, + timebase=timebase, + t=t, + selected=selected, + y=y, + fit=fit, + x_sel=x_sel, + own_ws=own_ws, + overall=overall, + n_refs=n_refs, + era5=era5, + ) + return MethodOutput(p50_overall=overall) + + def _factories(self) -> dict[str, Callable[[], Any]]: + """Nuisance/effect model factories with the configured LightGBM overrides applied.""" + params = self.model_params + return { + "make_outcome": lambda: make_outcome_model(**params), + "make_propensity": lambda: make_propensity_model(**params), + "make_effect": lambda: make_effect_model(**params), + } + + def _reference_mean_ws(self, features: pd.DataFrame, *, index: pd.DatetimeIndex) -> pd.Series: + """Mean wind speed across reference turbines (used only for ERA5 lag sync).""" + if self.wind_speed_col is None: + return pd.Series(np.nan, index=index) + ws_cols = [c for c in features.columns if c.startswith(f"{self.wind_speed_col}{QUALIFIER}")] + if not ws_cols: + return pd.Series(np.nan, index=index) + return features[ws_cols].mean(axis=1) + + def _own_ws(self, scada: pd.DataFrame, *, mi: MethodInput, index: pd.DatetimeIndex) -> np.ndarray | None: + """Return the test turbine's own wind speed aligned to ``index`` (for SCADA diagnostics), or None.""" + if self.wind_speed_col is None: + return None + test_rows = scada[scada[mi.turbine_col] == mi.test_wtg] + series = pd.Series( + test_rows[self.wind_speed_col].to_numpy(dtype=float), index=pd.DatetimeIndex(test_rows.index) + ) + return series[~series.index.duplicated()].reindex(index).to_numpy(dtype=float) + + def _add_era5( + self, features: pd.DataFrame, *, index: pd.DatetimeIndex, reference_ws: pd.Series, timebase: pd.Timedelta + ) -> tuple[pd.DataFrame, Any]: + """Sync ERA5 (if supplied) and append its features; return the (features, sync-result).""" + if self.era5_hourly_df is None: + return features, None + if self.wind_speed_col is None: + msg = "wind_speed_col is required to sync ERA5 (it provides the reference wind speed)." + raise ValueError(msg) + result = sync_era5(self.era5_hourly_df, target_index=index, reference_ws=reference_ws, timebase=timebase) + return pd.concat([features, era5_features(result.aligned)], axis=1), result + + def _select_rows( + self, scada: pd.DataFrame, *, mi: MethodInput, index: pd.DatetimeIndex, y: pd.Series, timebase: pd.Timedelta + ) -> np.ndarray: + """Boolean over ``index``: normally-operating test rows (cause-not-effect) with finite outcome.""" + test_rows = scada[scada[mi.turbine_col] == mi.test_wtg].sort_index() + keep = NormalOperationFilter( + active_power_col=self.active_power_col, + wind_speed_col=self.wind_speed_col, + availability_col=self.availability_col, + ).keep_mask(test_rows, timebase=timebase) + keep = keep.reindex(index, fill_value=False) + return keep.to_numpy() & np.isfinite(y.to_numpy(dtype=float)) + + def _write( + self, + mi: MethodInput, + *, + index: pd.DatetimeIndex, + timebase: pd.Timedelta, + t: pd.Series, + selected: np.ndarray, + y: pd.Series, + fit: Any, # noqa: ANN401 + x_sel: pd.DataFrame, + own_ws: np.ndarray | None, + overall: float, + n_refs: int, + era5: Any, # noqa: ANN401 + ) -> None: + """Assemble the diagnostic data and write the CSVs (+ plots), logging the top features.""" + upgrade_start = _upgrade_start(mi.upgrade_timing, index) + run_name = f"rlearner_{mi.test_wtg}_{upgrade_start:%Y%m%d}_{index.max():%Y%m%d}" + out_root = Path(self.out_dir) if self.out_dir is not None else Path(tempfile.mkdtemp(prefix="rlearner_")) + run_dir = out_root / run_name + run_dir.mkdir(parents=True, exist_ok=True) + ts = pd.Timestamp.utcnow().strftime("%Y%m%d_%H%M%S_%f") + y_arr = y.to_numpy(dtype=float) + data = diag.DiagnosticData( + test_wtg=mi.test_wtg, + mode="toggle" if isinstance(mi.upgrade_timing, ToggleSchedule) else "prepost", + index=index, + treated_all=t.to_numpy(), + selected_all=selected, + y_all=y_arr, + timebase=timebase, + tau=fit.tau, + m_hat=fit.m_hat, + e_hat=fit.e_hat, + mu0=fit.mu0, + y_selected=y_arr[selected], + condition_ws=own_ws[selected] if own_ws is not None else None, + condition_ws_label=f"{self.wind_speed_col} @ {mi.test_wtg}" if self.wind_speed_col is not None else None, + feature_names=list(x_sel.columns), + feature_values=x_sel, + outcome_model=fit.outcome_model, + propensity_model=fit.propensity_model, + effect_model=fit.effect_model, + overall_uplift=overall, + n_refs=n_refs, + era5_lag_rows=era5.best_lag_rows if era5 is not None else None, + era5_corr=era5.best_corr if era5 is not None else None, + era5_sweep=era5.sweep if era5 is not None else None, + ) + importance = diag.write_csvs(run_dir, run_name, ts, data) + diag.log_top_features(importance) + if self.save_plots: + diag.save_plots(run_dir / "plots", data, importance) + self._write_shared_diagnostics(mi, run_dir=run_dir, t=t, selected=selected, timebase=timebase, era5=era5) + + def _write_shared_diagnostics( + self, + mi: MethodInput, + *, + run_dir: Path, + t: pd.Series, + selected: np.ndarray, + timebase: pd.Timedelta, + era5: Any, # noqa: ANN401 + ) -> None: + """Emit the shared cross-method diagnostics (coverage/curves/histograms) and the run config.""" + # Align the schema's active-power / wind-speed roles to the columns this method was + # configured to read, so the shared plots use the right columns even if they differ. + columns = replace(self.columns, active_power=self.active_power_col) + if self.wind_speed_col is not None: + columns = replace(columns, wind_speed=self.wind_speed_col) + ctx = DiagnosticContext( + run_dir=run_dir, + test_wtg=mi.test_wtg, + turbine_col=mi.turbine_col, + columns=columns, + scada_df=mi.scada_df, + treated_ts=t.to_numpy().astype(bool), + used_ts=np.asarray(selected, dtype=bool), + timebase=timebase, + mode="toggle" if isinstance(mi.upgrade_timing, ToggleSchedule) else "prepost", + era5_df=era5.aligned if era5 is not None else None, + ) + write_common_diagnostics(ctx) + extra = { + "n_folds": self.n_folds, + "seed": self.seed, + "era5_lag_rows": era5.best_lag_rows if era5 is not None else None, + "era5_corr": era5.best_corr if era5 is not None else None, + } + write_run_config(ctx, method_name=self.name, method_params=self._config_params(), extra=extra) + + def _config_params(self) -> dict[str, Any]: + """Return the R-learner configuration recorded in the run-config YAML.""" + return { + "active_power_col": self.active_power_col, + "wind_speed_col": self.wind_speed_col, + "availability_col": self.availability_col, + "n_folds": self.n_folds, + "seed": self.seed, + "toggle_campaign_only": self.toggle_campaign_only, + "has_era5": self.era5_hourly_df is not None, + "model_params": self.model_params, + } + + +def _aggregate_uplift(*, tau: np.ndarray, mu0: np.ndarray, upgraded: np.ndarray) -> float: + """Overall uplift = sum(tau)/sum(mu0) over the upgraded rows (the ground-truth row set).""" + if not upgraded.any(): + return float("nan") + denom = float(np.sum(mu0[upgraded])) + if not np.isfinite(denom) or denom == 0: + return float("nan") + return float(np.sum(tau[upgraded]) / denom) diff --git a/benchmarking/baselines/rlearner/nuisance.py b/benchmarking/baselines/rlearner/nuisance.py new file mode 100644 index 00000000..e8e27b54 --- /dev/null +++ b/benchmarking/baselines/rlearner/nuisance.py @@ -0,0 +1,59 @@ +"""LightGBM nuisance/effect model factories for the R-learner. + +Three model roles (design note §4, Stage 1/2): + +* **outcome** ``m(x)=E[Y|X]`` — an L2 (or Huber) regressor. Energy is the integral of the + *mean*, so the energy-relevant model targets the mean, not the median (design note §2). +* **propensity** ``e(x)=E[T|X]`` — a classifier for the upgrade flag. +* **effect** ``tau(x)`` — a regressor fit on the R-learner pseudo-outcome. + +LightGBM is an optional dependency (the ``ml`` group); it is imported lazily so this package +imports without it installed. Defaults follow the design note's common hyperparameters; +callers (and tests) override via keyword arguments. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Any + +if TYPE_CHECKING: + from lightgbm import LGBMClassifier, LGBMRegressor + +# Design note §6 "common" hyperparameters; native NaN handling, seconds to train. +_COMMON: dict[str, Any] = { + "n_estimators": 600, + "learning_rate": 0.03, + "num_leaves": 63, + "min_child_samples": 200, + "subsample": 0.8, + "colsample_bytree": 0.8, + "verbose": -1, +} + + +def _import_lightgbm() -> Any: # noqa: ANN401 + """Import lightgbm lazily with a helpful error if the optional ``ml`` group is missing.""" + try: + import lightgbm # noqa: PLC0415 + except ImportError as exc: # pragma: no cover - exercised only without the optional dep + msg = "lightgbm is required for the R-learner method; install the 'ml' optional dependency group." + raise ImportError(msg) from exc + return lightgbm + + +def make_outcome_model(**overrides: Any) -> LGBMRegressor: # noqa: ANN401 + """L2 outcome regressor for ``E[Y|X]`` (the energy-relevant mean model).""" + lgb = _import_lightgbm() + return lgb.LGBMRegressor(objective="regression", **{**_COMMON, **overrides}) + + +def make_propensity_model(**overrides: Any) -> LGBMClassifier: # noqa: ANN401 + """Binary classifier for the propensity ``E[T|X]`` (flat ~0.5 for toggle, real for before/after).""" + lgb = _import_lightgbm() + return lgb.LGBMClassifier(objective="binary", **{**_COMMON, **overrides}) + + +def make_effect_model(**overrides: Any) -> LGBMRegressor: # noqa: ANN401 + """Regressor for the effect ``tau(x)``, fit on the R-learner pseudo-outcome.""" + lgb = _import_lightgbm() + return lgb.LGBMRegressor(objective="regression", **{**_COMMON, **overrides}) diff --git a/benchmarking/baselines/rlearner/rlearner.py b/benchmarking/baselines/rlearner/rlearner.py new file mode 100644 index 00000000..33f354c0 --- /dev/null +++ b/benchmarking/baselines/rlearner/rlearner.py @@ -0,0 +1,130 @@ +"""The pure cross-fit R-learner core (Robinson partialling-out). + +No I/O, no source vocabulary: given a feature matrix ``X`` (upgrade-invariant; NaN allowed), +outcome ``y`` and upgrade flag ``t``, it returns per-row ``tau`` (absolute effect), the +cross-fit nuisances ``m_hat``/``e_hat``, the baseline power ``mu0`` and the fitted +outcome/effect models for feature-importance diagnostics. + +Method (design note §4): + +1. Cross-fit the nuisances: over K folds, fit outcome ``m(x)=E[Y|X]`` and propensity + ``e(x)=E[T|X]`` on the training folds and predict on the held-out fold, so every row gets + an out-of-fold prediction it did not help train. Shuffled K-fold is valid because there are + no timestamp features. +2. Residualise: ``y_res = y - m_hat``, ``t_res = t - e_hat``. +3. Fit the effect model ``tau(x)`` on the pseudo-outcome ``y_res / t_res`` with R-loss weights + ``t_res**2`` (guarding the ``t_res≈0`` divide). +4. Baseline power ``mu0 = m_hat - e_hat * tau`` (no extra model). + +Collapses to plain regression adjustment when the propensity is flat (toggle) and +orthogonalises when it is not (before/after) — one code path for both modes. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING, Any + +import numpy as np + +if TYPE_CHECKING: + from collections.abc import Callable + + import numpy.typing as npt + import pandas as pd + +# Below this |t_res| the R-learner pseudo-outcome y_res/t_res is numerically unstable; such +# rows carry ~zero R-loss weight anyway, so they are dropped from the effect fit. +_MIN_T_RES = 1e-6 + + +def _import_kfold() -> Any: # noqa: ANN401 + """Import scikit-learn's ``KFold`` lazily with a helpful error if the optional ``ml`` group is missing. + + Mirrors the lazy LightGBM import so this package imports without the optional ``ml`` dependencies. + """ + try: + from sklearn.model_selection import KFold # noqa: PLC0415 + except ImportError as exc: # pragma: no cover - exercised only without the optional dep + msg = "scikit-learn is required for the R-learner method; install the 'ml' optional dependency group." + raise ImportError(msg) from exc + return KFold + + +@dataclass +class RLearnerFit: + """Per-row R-learner outputs plus the fitted models for diagnostics. + + :param tau: absolute per-row effect estimate ``tau(x)`` + :param m_hat: cross-fit outcome prediction ``E[Y|X]`` + :param e_hat: cross-fit propensity ``E[T|X]`` + :param mu0: baseline (un-upgraded) expected power ``m_hat - e_hat * tau`` + :param outcome_model: outcome model refit on all rows (for feature importance) + :param propensity_model: propensity model refit on all rows (for feature importance) + :param effect_model: the fitted effect model ``tau(x)`` (for feature importance) + """ + + tau: npt.NDArray[np.float64] + m_hat: npt.NDArray[np.float64] + e_hat: npt.NDArray[np.float64] + mu0: npt.NDArray[np.float64] + outcome_model: Any + propensity_model: Any + effect_model: Any + + +def cross_fit_rlearner( + x: pd.DataFrame, + *, + y: npt.ArrayLike, + t: npt.ArrayLike, + make_outcome: Callable[[], Any], + make_propensity: Callable[[], Any], + make_effect: Callable[[], Any], + n_folds: int = 5, + seed: int = 0, +) -> RLearnerFit: + """Cross-fit R-learner; returns per-row ``tau``/``m_hat``/``e_hat``/``mu0`` and fitted models. + + ``y`` and ``t`` must be finite (the caller filters downtime/NaN rows); ``x`` may contain NaN. + """ + y = np.asarray(y, dtype=float) + t = np.asarray(t, dtype=float) + n = len(y) + if not (np.isfinite(y).all() and np.isfinite(t).all()): + msg = "cross_fit_rlearner requires finite y and t; filter NaN/downtime rows before calling." + raise ValueError(msg) + + m_hat = np.empty(n) + e_hat = np.empty(n) + folds = _import_kfold()(n_splits=n_folds, shuffle=True, random_state=seed) + for train_idx, test_idx in folds.split(x): + x_tr, x_te = x.iloc[train_idx], x.iloc[test_idx] + m_hat[test_idx] = make_outcome().fit(x_tr, y[train_idx]).predict(x_te) + e_hat[test_idx] = make_propensity().fit(x_tr, t[train_idx]).predict_proba(x_te)[:, 1] + + y_res = y - m_hat + t_res = t - e_hat + weights = t_res**2 + usable = np.abs(t_res) >= _MIN_T_RES + pseudo = np.where(usable, y_res / np.where(usable, t_res, 1.0), 0.0) + + effect_model = make_effect() + effect_model.fit(x.iloc[usable], pseudo[usable], sample_weight=weights[usable]) + tau = effect_model.predict(x) + + # Refit outcome and propensity on all rows so feature importance reflects one model over the + # full data (the propensity importance shows which references reconstruct the treatment). + outcome_model = make_outcome().fit(x, y) + propensity_model = make_propensity().fit(x, t) + + mu0 = m_hat - e_hat * tau + return RLearnerFit( + tau=tau, + m_hat=m_hat, + e_hat=e_hat, + mu0=mu0, + outcome_model=outcome_model, + propensity_model=propensity_model, + effect_model=effect_model, + ) diff --git a/benchmarking/baselines/study_overnight_prepost.py b/benchmarking/baselines/study_overnight_prepost.py new file mode 100644 index 00000000..3ba0609c --- /dev/null +++ b/benchmarking/baselines/study_overnight_prepost.py @@ -0,0 +1,67 @@ +"""Longer (overnight) PREPOST study: oracle + naive + R-learner + v0 over seven upgrade profiles. + +Run on a server from the repo root:: + + uv run python -m benchmarking.baselines.study_overnight_prepost + +Outputs (per-profile leaderboards with uplift bias/spread/score and wall-time, the tidy +per-replicate results, per-method campaign-length curves, and each method's per-run +diagnostics) are written under ``WIND_UP_BENCHMARKING_OUTPUT_DIR`` (default +``~/temp/wind-up-benchmarking/prepost``). The first run downloads + caches the Hill of Towie +SCADA (Zenodo) and ERA5 (Open-Meteo, needs the ``era5`` group); install the ``ml`` group too +for the R-learner (lightgbm). + +Tune runtime vs precision with the two constants below. v0 (a full wind_up run per campaign) +and the R-learner are the cost; oracle and naive are ~free. +""" + +from __future__ import annotations + +import logging + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TREATMENT_START_RANGE, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, + MIN_PRE_MONTHS, + default_output_root, + run_prepost_study, +) +from benchmarking.baselines.overnight_common import start_overnight_run +from benchmarking.baselines.overnight_profiles import overnight_profiles +from benchmarking.harness import StudyConfig +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# --- tune these two for the runtime budget ------------------------------------------------- +N_REPLICATES = 4 # (turbine, treatment-start) draws per profile; the precision axis +CAMPAIGN_MONTHS = [1, 2, 3, 6, 12] # campaign-length sweep grid (matches study_power_model_compare, Issue 14) +# ------------------------------------------------------------------------------------------- + + +def main() -> None: + """Load real Hill of Towie SCADA and run the overnight prepost study (incl. v0).""" + study = StudyConfig( + mode="prepost", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=CAMPAIGN_MONTHS, + n_replicates=N_REPLICATES, + seed=0, + ) + out_dir = start_overnight_run("prepost", study, output_root=default_output_root().parent) + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + run_prepost_study(scada_df, profiles=overnight_profiles(), study=study, out_root=out_dir, include_v0=True) + + +if __name__ == "__main__": + main() diff --git a/benchmarking/baselines/study_overnight_toggle.py b/benchmarking/baselines/study_overnight_toggle.py new file mode 100644 index 00000000..0aa6ffe1 --- /dev/null +++ b/benchmarking/baselines/study_overnight_toggle.py @@ -0,0 +1,70 @@ +"""Longer (overnight) TOGGLE study: oracle + naive + R-learner + v0 over seven upgrade profiles. + +Run on a server from the repo root:: + + uv run python -m benchmarking.baselines.study_overnight_toggle + +Same seven profiles and outputs as the prepost study (see +:mod:`benchmarking.baselines.study_overnight_prepost`), but with a 20-min-on / 20-min-off +toggle. Outputs go under ``WIND_UP_BENCHMARKING_OUTPUT_DIR/toggle`` (default +``~/temp/wind-up-benchmarking/toggle``). The naive and R-learner methods fit the interleaved +on/off campaign window only (``toggle_campaign_only`` default), so on and off share a wind +distribution. + +Tune runtime vs precision with the two constants below. +""" + +from __future__ import annotations + +import logging + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TREATMENT_START_RANGE, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, + MIN_PRE_MONTHS, +) +from benchmarking.baselines.example_toggle_study import ( + DEFAULT_TOGGLE_PERIOD, + default_output_root, + run_toggle_study, +) +from benchmarking.baselines.overnight_common import start_overnight_run +from benchmarking.baselines.overnight_profiles import overnight_profiles +from benchmarking.harness import StudyConfig +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# --- tune these two for the runtime budget ------------------------------------------------- +N_REPLICATES = 4 # (turbine, treatment-start) draws per profile; the precision axis +CAMPAIGN_MONTHS = [1, 2, 3, 6, 12] # toggling-duration grid (matches compare grid, Issue 14; drops 9) +# ------------------------------------------------------------------------------------------- + + +def main() -> None: + """Load real Hill of Towie SCADA and run the overnight toggle study (incl. v0).""" + study = StudyConfig( + mode="toggle", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=CAMPAIGN_MONTHS, + toggle_period=DEFAULT_TOGGLE_PERIOD, + n_replicates=N_REPLICATES, + seed=0, + ) + out_dir = start_overnight_run("toggle", study, output_root=default_output_root().parent) + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + run_toggle_study(scada_df, profiles=overnight_profiles(), study=study, out_root=out_dir, include_v0=True) + + +if __name__ == "__main__": + main() diff --git a/benchmarking/baselines/study_power_model_compare.py b/benchmarking/baselines/study_power_model_compare.py new file mode 100644 index 00000000..376112e8 --- /dev/null +++ b/benchmarking/baselines/study_power_model_compare.py @@ -0,0 +1,926 @@ +"""Re-run ``power_model`` over the overnight cases and compare it against the existing v0 + naive. + +Iterative-development helper. The overnight v0 runs are very slow, so they are computed once +(:mod:`benchmarking.baselines.study_overnight_prepost` / ``study_overnight_toggle`` with +``include_v0=True``) and kept on disk. As ``power_model`` is improved you only want to re-run the +cheap methods and re-draw the comparison plots, reusing the frozen v0 numbers. That is what this +script does: + +1. **Run** ``power_model`` **and** ``naive_ratio`` over the *exact* current overnight prepost + toggle + configs — same seven :func:`~benchmarking.baselines.overnight_profiles.overnight_profiles`, same + campaign grids (Issue 14: 1/2/3/6/12 months both modes), ``n_replicates=4``, ``seed=0``. naive is + very cheap (no wind_up pipeline) so it is recomputed fresh here rather than pulled from the reference + run — that keeps naive on current code and independent of the reference. Only v0 is *not* recomputed + (it is very slow and already frozen in the reference run, which now covers the full 1-12-month grid). +2. **Merge** each fresh ``power_model`` + ``naive_ratio`` result with the ``v0_binned`` rows pulled + from the reference overnight directory. +3. **Plot** a three-method campaign-length comparison (``naive_ratio``, ``v0_binned``, + ``power_model``) per profile, and write merged per-profile + all-profiles leaderboards. + +An alignment guard checks that the method-independent ground ``truth`` of the fresh cases equals the +reference run's truth on the campaign lengths they share; a mismatch there means the configs have +drifted and the merge would be comparing different cases, so it fails loudly. The current reference +covers the full 1-12-month grid, so every fresh case normally has a reference match; any fresh case a +reference does not cover is expected and skipped by the guard. + +**Benchmark regression tracking.** ``power_model``'s bias/spread/score per +``(mode, profile, campaign_months)`` at a known-good commit is frozen in a committed JSON benchmark +(``study_power_model_compare_baseline.json``, next to this script). Every run diffs the fresh +``power_model`` against it and logs a mean-over-profiles table of the deltas (spread/score: a +negative delta is an improvement; bias: a smaller ``|bias|`` is better) plus a per-cell +``benchmark_comparison_.csv``, so an attempt to improve ``power_model`` is scored objectively +against where it stands today. + +**This benchmark is machine-specific: record and diff it on one machine only.** Every cell here is +``power_model``, and LightGBM's threaded float reduction order depends on the machine, so a benchmark +recorded elsewhere reports a permanent false MOVED of ~0.7 pp — 14x the same-machine noise (F30). +Unlike ``study_toggle_methods_compare``, which splits portable from machine-specific cells across +files, this script keeps one file because it has no portable cells to split off. The committed +benchmark is recorded on the Linux laptop; run it there. If that ever stops being true, the +portable/per-platform mechanism in ``study_toggle_methods_compare`` is the thing to copy. + +**Accepting an improvement without a second sweep.** Every full sweep also drops a *candidate* +baseline (``/candidate_baseline.json``, seeded from the committed baseline so unrun modes +are preserved) — i.e. exactly what the committed file would become if you accept this run. If the run +looks good, promote it near-instantly with ``--accept-candidate`` (copies the candidate over the +committed JSON and exits, no re-run); then commit the JSON. ``--update-baseline`` still records +directly from a fresh sweep, but with the candidate flow you rarely need it. + +**Per-bin before/after view.** Alongside the overall diff, the run also surfaces the *conditional* +before/after for the change under test on the two condition-dependent hard cases plus the placebo +(:data:`COVERED_PROFILES`): a per-bin ``|bias|`` table with a ``better``/``worse``/``~`` verdict +(``conditional_benchmark_comparison_.csv`` + log) and one overlay per ``(profile, condition)`` +plotting truth vs the benchmark vs the current run +(``conditional_before_after__.png``). The benchmark's per-bin curve is +reconstructed from its stored per-bin bias, so no benchmark-JSON change is needed. Both the overall +tally and the per-bin verdict use one materiality band (:data:`_MATERIAL_PP`). + +Run from the repo root:: + + uv run python -m benchmarking.baselines.study_power_model_compare \ + --reference-dir "~/temp/wind-up-benchmarking/badass overnight 20260708" + +For fast feedback on a power_model change, restrict to one mode and one profile — e.g. +``--modes prepost --profiles cp_0pct`` fits a single case in ~minutes (vs ~30 for the full sweep) and +still emits its overall + per-bin before/after view. Use ``--skip-run`` to only re-merge/re-plot (and +re-diff the benchmark) from a previous ``power_model`` run under ``--output-dir`` (e.g. to tweak +plotting without re-fitting). Use ``--update-baseline`` to re-record the benchmark from the current +run (needs the full profile set — it cannot be combined with ``--profiles``), or ``--accept-candidate`` +to promote the candidate a previous full sweep already wrote (no re-run). +""" + +from __future__ import annotations + +import argparse +import dataclasses +import json +import logging +import subprocess +from datetime import datetime, timezone +from functools import partial +from pathlib import Path +from typing import Any + +import numpy as np +import pandas as pd + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TREATMENT_START_RANGE, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, + MIN_PRE_MONTHS, + save_per_method_curve, +) +from benchmarking.baselines.example_toggle_study import DEFAULT_TOGGLE_PERIOD +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.naive_ratio import NaiveRatioMethod +from benchmarking.baselines.overnight_profiles import overnight_profiles +from benchmarking.baselines.power_model import PowerModelMethod +from benchmarking.harness import ( + StudyConfig, + conditional_leaderboard, + leaderboard, + plot_campaign_curves, + plot_conditional_uplift, + score_study, +) +from benchmarking.synthetic import HOT_COLUMNS, HOT_RATED_POWER_KW +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# The three methods compared in the merged plots (oracle is dropped: it is only a truth anchor). +COMPARE_METHODS = ["naive_ratio", "v0_binned", "power_model"] +REUSED_METHODS = ["v0_binned"] # only v0 is reused from the reference run; naive is recomputed fresh + +# Issue 14: the out-of-the-box method is scored across the full 1-12-month range in both modes, so the +# short-campaign regime (F16) is part of every A/B. The current reference run freezes v0 across the whole +# 1-12-month grid; naive is recomputed fresh here (current code) rather than reused from the reference. +PREPOST_CAMPAIGN_MONTHS = [1, 2, 3, 6, 12] +TOGGLE_CAMPAIGN_MONTHS = [1, 2, 3, 6, 12] +N_REPLICATES = 4 +SEED = 0 + +# Keys that identify one scored case independently of the method. +_CASE_KEYS = ["profile", "test_wtg", "campaign_months", "treatment_start"] +_DEFAULT_REFERENCE_DIR = Path.home() / "temp" / "wind-up-benchmarking" / "badass overnight 20260708" +_DEFAULT_OUTPUT_DIR = Path.home() / "temp" / "wind-up-benchmarking" / "power_model_compare" + +# The committed power_model benchmark: its bias/spread/score per (mode, profile, campaign) frozen at +# a known-good commit, so future power_model changes are scored against it. Lives next to this script +# (tracked) — update it deliberately with --update-baseline when an improvement is accepted. +_BASELINE_PATH = Path(__file__).resolve().parent / "study_power_model_compare_baseline.json" +_BASELINE_SCHEMA = "power_model_compare_baseline_v2" +# Per-cell metrics recorded and diffed. spread/score: lower is better; bias: |bias| nearer 0 is better. +_METRIC_COLS = ["bias", "spread", "score"] +# Materiality band (percentage points) for the "did this change help/hurt/neutral" verdicts, shared by +# the overall tally() and the per-bin conditional table so the report speaks one language. A move whose +# magnitude is <= this reads neutral ("~"). 0.1 pp is well above floating-point noise (so an identical +# deterministic re-run still reads all-neutral) yet small enough to catch any change worth judging; +# hard-case per-bin biases run to tens of pp. tally() works on fractional deltas, so it uses the +# fractional form _MATERIAL_PP / 100. +_MATERIAL_PP = 0.1 +_PP = 100.0 # fraction -> percentage points +# Profiles that get the per-bin before/after conditional view: the two condition-dependent hard cases +# plus the placebo (true uplift 0 in every bin — confirms a change adds no per-bin bias). The overall +# benchmark diff already covers all profiles; the other homogeneous cp_* have flat per-bin truth. +COVERED_PROFILES = ("cp_0pct", "ti_dependent_cp", "ws_dependent_cp") + + +def _prepost_study() -> StudyConfig: + return StudyConfig( + mode="prepost", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=PREPOST_CAMPAIGN_MONTHS, + n_replicates=N_REPLICATES, + seed=SEED, + ) + + +def _toggle_study() -> StudyConfig: + return StudyConfig( + mode="toggle", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_months=TOGGLE_CAMPAIGN_MONTHS, + toggle_period=DEFAULT_TOGGLE_PERIOD, + n_replicates=N_REPLICATES, + seed=SEED, + ) + + +def _select_profiles(requested: list[str] | None) -> dict[str, list]: + """Return the overnight profiles to score: all when ``requested`` is ``None``, else the named subset. + + A subset (e.g. ``["cp_0pct"]``) gives fast iteration on a power_model change — one case in ~minutes + rather than the whole ~30-minute sweep. Unknown names fail loudly rather than silently scoring less. + """ + all_profiles = overnight_profiles() + if requested is None: + return all_profiles + unknown = [name for name in requested if name not in all_profiles] + if unknown: + msg = f"unknown profile(s) {unknown}; available: {sorted(all_profiles)}" + raise ValueError(msg) + return {name: all_profiles[name] for name in requested} + + +def _make_power_model( + out_dir: Path, *, era5_hourly_df: pd.DataFrame, overrides: dict[str, Any] | None = None +) -> PowerModelMethod: + """Construct the HoT-configured ``power_model`` method (its default: conditional uplift on). + + ``overrides`` are ``PowerModelMethod`` field overrides for candidate A/B runs (the Issue 9/10/11 + protocol), e.g. ``{"matching_vars": ["wind_speed_100m"]}``; unknown field names fail loudly and + JSON lists are coerced to the tuples the dataclass fields expect. + """ + # Only data-schema description is passed here now: the accepted behaviour defaults (F13 + # availability_feature/era5_exclude, F14 min_child_samples) live on the PowerModelMethod class + # (Issue 14), so a bare method already *is* the benchmarked config. The Issue 11 / F12 reference + # active-power minimum is carried by HOT_COLUMNS' active_power_min role, so no specialist config + # is needed here either. + kwargs: dict[str, Any] = { + "columns": HOT_COLUMNS, + "baseline_rated_power_kw": HOT_RATED_POWER_KW, + "era5_hourly_df": era5_hourly_df, + "out_dir": out_dir / "power_model_runs", + } + if overrides: + known = {f.name for f in dataclasses.fields(PowerModelMethod)} + unknown = sorted(set(overrides) - known) + if unknown: + msg = f"unknown PowerModelMethod field(s) in --method-overrides: {unknown}" + raise ValueError(msg) + kwargs.update({k: tuple(v) if isinstance(v, list) else v for k, v in overrides.items()}) + return PowerModelMethod(**kwargs) + + +def run_power_model( + mode: str, + out_dir: Path, + *, + profiles: list[str] | None = None, + method_overrides: dict[str, Any] | None = None, +) -> pd.DataFrame: + """Score ``power_model`` **and** ``naive_ratio`` over the overnight cases for one mode (no v0/oracle). + + ``profiles`` restricts to a subset of :func:`overnight_profiles` (default: all) for fast feedback on + a power_model change. power_model runs at its default (conditional uplift on), so every case also + emits the per-bin conditional cells the benchmark tracks. naive is recomputed fresh here (it is very + cheap and has no wind_up pipeline) so the merge does not depend on the reference run's naive rows and + naive always reflects current code. Each fresh row still + carries the harness's method-independent ground ``truth`` (so the merge's alignment guard has its + cross-check against v0). Writes a per-profile ``results_*.csv`` and per-profile method curves under + ``out_dir``, and returns the concatenated tidy results. + """ + out_dir.mkdir(parents=True, exist_ok=True) + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + context = build_hot_v0_context(wtg_names=DEFAULT_TURBINE_SUBSET) + study = _prepost_study() if mode == "prepost" else _toggle_study() + + all_results = [] + for profile_name, profile in _select_profiles(profiles).items(): + # Per-profile subfolder: the method's run dir is power_model___ (no profile), so + # profiles sharing a (wtg, window) would otherwise overwrite each other's non-timestamped diagnostics. + method = _make_power_model( + out_dir / profile_name, + era5_hourly_df=context.reanalysis_datasets[0].data, + overrides=method_overrides, + ) + naive = NaiveRatioMethod( + columns=HOT_COLUMNS, + out_dir=out_dir / profile_name / "naive_runs", + ) + logger.info("Scoring %s profile %s with power_model + naive_ratio", mode, profile_name) + results = score_study( + scada_df, + profile=profile, + methods=[naive, method], + study=study, + profile_name=profile_name, + on_method_complete=partial(save_per_method_curve, out_dir, profile_name), + ) + results.to_csv(out_dir / f"results_{profile_name}.csv", index=False) + all_results.append(results) + return pd.concat(all_results, ignore_index=True) + + +def _load_fresh_results(mode_out_dir: Path) -> pd.DataFrame: + """Concatenate the per-profile ``results_*.csv`` a previous run wrote (for ``--skip-run``).""" + files = sorted(mode_out_dir.glob("results_*.csv")) + if not files: + msg = f"no results_*.csv under {mode_out_dir}; run without --skip-run first." + raise FileNotFoundError(msg) + return pd.concat([pd.read_csv(f) for f in files], ignore_index=True) + + +def _resolve_results_dir(reference_mode_dir: Path) -> Path: + """Return the directory that actually holds the reference ``results_*.csv`` for a mode. + + ``study_overnight_prepost`` / ``study_overnight_toggle`` (via ``start_overnight_run``) write each run + into a timestamped subdir ``//``, so a reference dir points at ``//`` + with one such subdir inside. Accept either layout: the ``results_*.csv`` directly under + ``reference_mode_dir`` (flat, as older hand-assembled references were), or a single timestamped subdir + holding them. Two or more candidate subdirs is ambiguous (which run?) and fails loudly. + """ + if list(reference_mode_dir.glob("results_*.csv")): + return reference_mode_dir + candidates = sorted(d for d in reference_mode_dir.iterdir() if d.is_dir() and list(d.glob("results_*.csv"))) + if len(candidates) == 1: + return candidates[0] + if not candidates: + msg = f"no results_*.csv under {reference_mode_dir} or a single timestamped subdir of it" + raise FileNotFoundError(msg) + names = [d.name for d in candidates] + msg = ( + f"multiple run subdirs with results_*.csv under {reference_mode_dir} ({names}); " + f"point --reference-dir at one run" + ) + raise ValueError(msg) + + +def _load_reference_methods(reference_mode_dir: Path, profiles: list[str], methods: list[str]) -> pd.DataFrame: + """Load the requested methods' rows for the given profiles from the reference run directory.""" + results_dir = _resolve_results_dir(reference_mode_dir) + frames = [] + for profile in profiles: + path = results_dir / f"results_{profile}.csv" + if not path.exists(): + msg = f"reference results missing for profile {profile!r}: {path}" + raise FileNotFoundError(msg) + df = pd.read_csv(path) + frames.append(df[df["method"].isin(methods)]) + return pd.concat(frames, ignore_index=True) + + +def _case_key(df: pd.DataFrame) -> pd.Series: + """Build a stable per-case string key (treatment_start normalised so str/Timestamp compare equal).""" + ts = pd.to_datetime(df["treatment_start"], utc=True).dt.strftime("%Y-%m-%d %H:%M:%S%z") + return ( + df["profile"].astype(str) + + "|" + + df["test_wtg"].astype(str) + + "|" + + df["campaign_months"].astype(int).astype(str) + + "|" + + ts + ) + + +def _check_alignment(fresh: pd.DataFrame, reference: pd.DataFrame) -> None: + """Fail loudly if the fresh and reference runs disagree on ground truth where they overlap. + + Ground ``truth`` is method-independent and deterministic in the study config + seed, so equal + keys must carry equal truth. The current reference run covers the full 1-12-month grid, so the fresh + cases normally all have a reference match; any fresh case a reference does not cover is expected and + not an error — only the fresh ∩ reference intersection is truth-checked; a mismatch *there* + still means the configs drifted and the merge would compare different cases. + """ + f_overall = fresh[fresh["condition"] == "overall"].copy() + r_overall = reference[reference["condition"] == "overall"].copy() + f_truth = f_overall.assign(key=_case_key(f_overall)).groupby("key")["truth"].first() + r_truth = r_overall.assign(key=_case_key(r_overall)).groupby("key")["truth"].first() + + common = f_truth.index.intersection(r_truth.index) + bad = ~np.isclose(f_truth.loc[common].to_numpy(), r_truth.loc[common].to_numpy(), rtol=1e-6, atol=1e-9) + if bad.any(): + example = common[bad][0] + msg = ( + f"{int(bad.sum())} case(s) disagree on ground truth between the fresh power_model run and the " + f"reference run (e.g. {example}: fresh={f_truth[example]:.6g} vs ref={r_truth[example]:.6g}). " + f"The overnight config has drifted from the reference run; the merge would compare different cases." + ) + raise ValueError(msg) + logger.info("Alignment OK: %d cases match the reference run on ground truth.", len(common)) + + +def _git_commit() -> str: + """Return the short HEAD commit (``-dirty`` if *tracked* files are modified), or ``unknown``. + + ``--untracked-files=no`` matches ``study_toggle_methods_compare`` and is the point: only tracked + modifications make a run irreproducible from its commit. Counting untracked files (a scratch + script, a local CLAUDE.md) made ``--update-baseline`` impossible for anyone with a stray file, and + is the likeliest reason the committed baseline is stamped ``e2e21b0-dirty`` despite reproducing + exactly (F30). + """ + repo = Path(__file__).resolve().parent + try: + commit = subprocess.run( + ["git", "rev-parse", "--short", "HEAD"], # noqa: S607 + cwd=repo, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + dirty = subprocess.run( + ["git", "status", "--porcelain", "--untracked-files=no"], # noqa: S607 + cwd=repo, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + except (subprocess.SubprocessError, OSError): + return "unknown" + return f"{commit}-dirty" if dirty else commit + + +def power_model_leaderboard(fresh: pd.DataFrame) -> pd.DataFrame: + """One row per (profile, campaign, condition, bin) of power_model bias/spread/score (overall incl.).""" + pm = fresh[fresh["method"] == "power_model"] + overall = leaderboard(pm).assign(condition="overall", condition_bin="overall") + conditional = conditional_leaderboard(pm) + cols = ["profile", "campaign_months", "condition", "condition_bin", *_METRIC_COLS] + stacked = pd.concat([overall[cols], conditional[cols]], ignore_index=True) + return stacked.sort_values(["profile", "campaign_months", "condition", "condition_bin"]) + + +def record_baseline( + lb_by_mode: dict[str, pd.DataFrame], + study_by_mode: dict[str, StudyConfig], + path: Path, + *, + seed_path: Path | None = None, + git_commit: str | None = None, +) -> None: + """Write/refresh the benchmark for the given modes (other modes in the file are kept). + + Sibling modes not in ``lb_by_mode`` are inherited from ``seed_path`` (default: ``path`` itself). A + *candidate* baseline is written by pointing ``path`` at the candidate file while seeding from the + committed baseline, so the candidate equals what accepting this run would make the committed file. + + ``git_commit`` should be captured *before* the sweep: this one takes hours, so reading HEAD here + would stamp whatever was committed meanwhile rather than the code that produced the numbers. + """ + seed = seed_path if seed_path is not None else path + doc: dict[str, Any] = {"schema": _BASELINE_SCHEMA, "modes": {}} + if seed.exists(): + loaded = json.loads(seed.read_text()) + if loaded.get("schema") == _BASELINE_SCHEMA: + doc = loaded + doc.setdefault("modes", {}) + commit = _git_commit() if git_commit is None else git_commit + now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") + for mode, lb in lb_by_mode.items(): + study = study_by_mode[mode] + doc["modes"][mode] = { + "recorded_utc": now, + "git_commit": commit, + "n_replicates": study.n_replicates, + "seed": study.seed, + "campaign_months": list(study.campaign_lengths), # months-only study; see _prepost_study/_toggle_study + "profiles": sorted(lb["profile"].unique()), + "cells": lb.round(8).to_dict(orient="records"), + } + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(doc, indent=2) + "\n") + logger.info("Recorded power_model benchmark for %s at %s (commit %s)", list(lb_by_mode), path, commit) + + +def _refuse_dirty_update(parser: argparse.ArgumentParser, *, git_commit: str, update_baseline: bool) -> None: + """Refuse to record the committed benchmark from a dirty tree. + + A committed benchmark is only worth anything if a reader can check out its commit and reproduce + it. Without ``--update-baseline`` a dirty tree is fine — that run only reports. + """ + if update_baseline and git_commit.endswith("-dirty"): + parser.error( + f"refusing to --update-baseline from a dirty tree (commit {git_commit}): the committed " + f"benchmark must be reproducible from its commit. Commit your changes first, then re-run." + ) + + +def _candidate_path(output_dir: Path) -> Path: + """Where a full sweep drops its candidate baseline (promote it with ``--accept-candidate``).""" + return output_dir / "candidate_baseline.json" + + +def accept_candidate(candidate_path: Path, baseline_path: Path) -> None: + """Promote a candidate baseline to the committed benchmark — a near-instant, no-sweep action. + + Every full sweep writes a candidate (see :func:`record_baseline`); this copies it over + ``baseline_path`` so accepting an improvement never means re-running the ~30-minute sweep. + """ + if not candidate_path.exists(): + msg = f"no candidate baseline at {candidate_path}; run a full sweep (all profiles) first to produce one." + raise FileNotFoundError(msg) + text = candidate_path.read_text() + doc = json.loads(text) + schema = doc.get("schema") + if schema != _BASELINE_SCHEMA: + msg = f"candidate {candidate_path} has schema {schema!r}, expected {_BASELINE_SCHEMA!r}; regenerate it." + raise ValueError(msg) + # The sweep that produced the candidate may have run from a dirty tree; promoting it would commit a + # benchmark no commit reproduces. This is the likeliest way a `-dirty` baseline gets committed, + # since accepting is the documented no-re-run path. + dirty = { + mode: m["git_commit"] + for mode, m in doc.get("modes", {}).items() + if str(m.get("git_commit", "")).endswith("-dirty") + } + if dirty: + msg = ( + f"refusing to promote {candidate_path}: it was recorded from a dirty tree ({dirty}). The committed " + f"benchmark must be reproducible from its commit. Commit your changes and re-run the sweep first." + ) + raise ValueError(msg) + baseline_path.parent.mkdir(parents=True, exist_ok=True) + baseline_path.write_text(text) + logger.info( + "Promoted candidate %s -> committed baseline %s (modes %s). Commit the new JSON.", + candidate_path, + baseline_path, + sorted(json.loads(text).get("modes", {})), + ) + + +def _load_baseline_cells(mode: str, path: Path) -> tuple[pd.DataFrame, dict[str, Any]] | None: + """Load the recorded benchmark cells (+ provenance) for ``mode``, or ``None`` if not recorded.""" + if not path.exists(): + return None + doc = json.loads(path.read_text()) + if doc.get("schema") != _BASELINE_SCHEMA: + logger.warning( + "Baseline %s has schema %r, expected %r — run --update-baseline to regenerate.", + path, + doc.get("schema"), + _BASELINE_SCHEMA, + ) + return None + entry = doc.get("modes", {}).get(mode) + if entry is None: + return None + return pd.DataFrame(entry["cells"]), entry + + +def _tally(delta: pd.Series, n_cells: int, *, threshold: float) -> str: + """Count cells that moved beyond the materiality band: ``better`` (down) vs ``worse`` (up). + + ``delta`` is a fractional metric change; ``threshold`` is the band in the same fractional units. + """ + better = int((delta < -threshold).sum()) + worse = int((delta > threshold).sum()) + return f"{better} better / {worse} worse (of {n_cells})" + + +def _verdict(d_abs_bias_pp: float, material_pp: float) -> str: + """``better`` / ``worse`` / ``~`` for a per-bin |bias| change (pp) against the materiality band.""" + if d_abs_bias_pp < -material_pp: + return "better" + if d_abs_bias_pp > material_pp: + return "worse" + return "~" + + +_MERGE_KEYS = ["profile", "campaign_months", "condition", "condition_bin"] + + +def _covered_longest(fresh_cond_lb: pd.DataFrame) -> pd.DataFrame: + """Fresh conditional rows restricted to the covered profiles' per-condition bins at their longest campaign.""" + fresh = fresh_cond_lb[ + fresh_cond_lb["profile"].isin(COVERED_PROFILES) & (fresh_cond_lb["condition"] != "overall") + ].copy() + if fresh.empty: + return fresh + longest = fresh.groupby("profile")["campaign_months"].transform("max") + return fresh[fresh["campaign_months"] == longest] + + +def conditional_before_after_table( + fresh_cond_lb: pd.DataFrame, baseline_cells: pd.DataFrame, *, material_pp: float = _MATERIAL_PP +) -> pd.DataFrame: + """Per-bin before/after table for the covered profiles at their longest campaign (values in pp). + + ``fresh_cond_lb`` is a fresh :func:`~benchmarking.harness.conditional_leaderboard` (carrying + ``mean_estimate`` / ``mean_truth`` / ``bias`` per bin); ``baseline_cells`` is the recorded benchmark + (per-bin ``bias`` only). The benchmark's per-bin estimate is reconstructed exactly as + ``mean_truth + bias`` (ground truth is deterministic and alignment-guarded), so no benchmark-JSON + schema change is needed. Returns one row per ``(profile, condition, condition_bin)`` with the + reconstructed ``est_before``, fresh ``est_after``, ``|bias|`` before/after, their pp delta, and a + ``better``/``worse``/``~`` verdict against ``material_pp``. + """ + columns = [ + "profile", "condition", "condition_bin", "campaign_months", + "mean_truth", "est_before", "est_after", "abs_bias_before", "abs_bias_after", "d_abs_bias", "verdict", + ] # fmt: skip + fresh = _covered_longest(fresh_cond_lb) + if fresh.empty: + return pd.DataFrame(columns=columns) + + base = baseline_cells[[*_MERGE_KEYS, "bias"]].rename(columns={"bias": "bias_before"}) + merged = fresh.merge(base, on=_MERGE_KEYS, how="inner") + + out = pd.DataFrame( + { + "profile": merged["profile"], + "condition": merged["condition"], + "condition_bin": merged["condition_bin"], + "campaign_months": merged["campaign_months"], + "mean_truth": merged["mean_truth"] * _PP, + "est_before": (merged["mean_truth"] + merged["bias_before"]) * _PP, + "est_after": merged["mean_estimate"] * _PP, + "abs_bias_before": merged["bias_before"].abs() * _PP, + "abs_bias_after": merged["bias"].abs() * _PP, + } + ) + out["d_abs_bias"] = out["abs_bias_after"] - out["abs_bias_before"] + out["verdict"] = out["d_abs_bias"].apply(lambda d: _verdict(d, material_pp)) + return out.sort_values(["profile", "condition", "condition_bin"]).reset_index(drop=True) + + +def _overlay_frame(fresh_cond_lb: pd.DataFrame, baseline_cells: pd.DataFrame) -> pd.DataFrame: + """Two-"method" frame (benchmark + current) for :func:`plot_conditional_uplift`, in fractional units. + + The benchmark series' ``mean_estimate`` is reconstructed as ``mean_truth + bias`` and carries the + benchmark's own per-bin ``spread``; the current series is the fresh estimate/spread. Both share the + fresh ``mean_truth`` (the truth line). Empty if no covered-profile cells line up. + """ + fresh = _covered_longest(fresh_cond_lb) + plot_cols = ["profile", "condition", "condition_bin", "method", "mean_truth", "mean_estimate", "spread"] + if fresh.empty: + return pd.DataFrame(columns=plot_cols) + base = baseline_cells[[*_MERGE_KEYS, "bias", "spread"]].rename( + columns={"bias": "bias_before", "spread": "spread_before"} + ) + merged = fresh.merge(base, on=_MERGE_KEYS, how="inner") + shared = {c: merged[c] for c in ["profile", "condition", "condition_bin", "mean_truth"]} + benchmark = pd.DataFrame( + {**shared, "method": "power_model (benchmark)", + "mean_estimate": merged["mean_truth"] + merged["bias_before"], "spread": merged["spread_before"]} + ) # fmt: skip + current = pd.DataFrame( + {**shared, "method": "power_model (current)", + "mean_estimate": merged["mean_estimate"], "spread": merged["spread"]} + ) # fmt: skip + return pd.concat([benchmark, current], ignore_index=True)[plot_cols] + + +def conditional_before_after(mode: str, fresh: pd.DataFrame, baseline_path: Path, comparison_dir: Path) -> None: + """Per-bin before/after view for the covered profiles: a delta table (CSV + log) and overlay plots. + + Reads the recorded benchmark, computes the fresh power_model conditional leaderboard, and writes + ``conditional_benchmark_comparison_.csv`` plus one + ``conditional_before_after__.png`` overlay (truth vs benchmark vs current) per + covered ``(profile, condition)``. A no-op (with a warning) when no benchmark is recorded for ``mode``. + """ + loaded = _load_baseline_cells(mode, baseline_path) + if loaded is None: + logger.warning( + "No power_model benchmark for %s in %s — skipping the per-bin before/after view.", + mode, + baseline_path, + ) + return + base, prov = loaded + cond_lb = conditional_leaderboard(fresh[fresh["method"] == "power_model"]) + + table = conditional_before_after_table(cond_lb, base) + if table.empty: + logger.info("%s: no covered-profile conditional cells line up with the benchmark; nothing to show.", mode) + return + table.round(4).to_csv(comparison_dir / f"conditional_benchmark_comparison_{mode}.csv", index=False) + logger.info( + "%s power_model per-bin |bias| vs benchmark (recorded %s, commit %s) [pp], verdict band +/-%.3g pp:\n%s", + mode, + prov.get("recorded_utc", "?"), + prov.get("git_commit", "?"), + _MATERIAL_PP, + table.round(3).to_string(index=False), + ) + + overlay = _overlay_frame(cond_lb, base) + commit = prov.get("git_commit", "?") + for (profile, condition), subset in overlay.groupby(["profile", "condition"]): + plot_conditional_uplift( + subset, + condition=condition, + save_path=comparison_dir / f"conditional_before_after_{profile}_{condition}.png", + title=f"{mode} - {profile} power_model benchmark vs current vs {condition} (benchmark {commit})", + ) + + +def compare_to_benchmark(mode: str, lb: pd.DataFrame, baseline_path: Path, comparison_dir: Path) -> None: + """Diff the fresh power_model bias/spread/score against the committed benchmark and report it. + + Writes a per-cell ``benchmark_comparison_.csv`` and logs a mean-over-profiles table with the + deltas. spread/score: a negative delta is an improvement; bias: a smaller ``|bias|`` is better. + """ + loaded = _load_baseline_cells(mode, baseline_path) + if loaded is None: + logger.warning( + "No power_model benchmark recorded for %s in %s yet — run with --update-baseline to set it.", + mode, + baseline_path, + ) + return + base, prov = loaded + base = base[base["profile"].isin(lb["profile"].unique())] # scope to the profiles actually run + merge_keys = ["profile", "campaign_months", "condition", "condition_bin"] + merged = lb.merge(base, on=merge_keys, how="outer", suffixes=("", "_base")) + for col in _METRIC_COLS: + merged[f"d_{col}"] = merged[col] - merged[f"{col}_base"] + merged["d_abs_bias"] = merged["bias"].abs() - merged["bias_base"].abs() + merged.sort_values(["profile", "campaign_months", "condition", "condition_bin"]).to_csv( + comparison_dir / f"benchmark_comparison_{mode}.csv", index=False + ) + + report = merged.groupby(["campaign_months", "condition"]).mean(numeric_only=True) + overall = merged.mean(numeric_only=True).to_frame().T + overall.index = ["ALL"] + table = pd.concat([report, overall]) + show = pd.DataFrame( + { + "bias": table["bias"] * _PP, + "Δbias": table["d_bias"] * _PP, + "spread": table["spread"] * _PP, + "Δspread": table["d_spread"] * _PP, + "score": table["score"] * _PP, + "Δscore": table["d_score"] * _PP, + } + ).round(3) + # Only rows where both fresh and benchmark metrics are present can contribute to the tallies + # (outer-merge leaves NaN deltas otherwise), so count those to keep "(of N)" consistent. + metric_cols = _METRIC_COLS + [f"{col}_base" for col in _METRIC_COLS] + n_cells = int(merged[metric_cols].notna().all(axis=1).sum()) + threshold = _MATERIAL_PP / _PP # tally works on fractional deltas; the band is defined in pp. + + logger.info( + "%s power_model vs benchmark (recorded %s, commit %s); mean over profiles [pp], " + "Δ<0 = better for spread/score (|Δ| <= %.3g pp reads neutral):\n%s\n" + "cells: spread %s; score %s; |bias| %s", + mode, + prov.get("recorded_utc", "?"), + prov.get("git_commit", "?"), + _MATERIAL_PP, + show.to_string(), + _tally(merged["d_spread"], n_cells, threshold=threshold), + _tally(merged["d_score"], n_cells, threshold=threshold), + _tally(merged["d_abs_bias"], n_cells, threshold=threshold), + ) + + +def _conditional_plot_subset(cond_lb: pd.DataFrame, profile: str, condition: str) -> pd.DataFrame: + """Rows for one (profile, condition) at the longest campaign length only. + + ``conditional_leaderboard`` keeps one row per ``(method, profile, campaign_months, condition, + condition_bin)``, but :func:`~benchmarking.harness.plots.plot_conditional_uplift` expects a + single row per ``condition_bin`` per method. Collapse to the longest campaign (most data, so the + cleanest per-bin estimate) rather than mixing campaign lengths into one line. + """ + subset = cond_lb[(cond_lb["profile"] == profile) & (cond_lb["condition"] == condition)] + if subset.empty: + return subset + longest = int(subset["campaign_months"].max()) + return subset[subset["campaign_months"] == longest] + + +def merge_and_plot(mode: str, fresh: pd.DataFrame, reference_mode_dir: Path, out_dir: Path) -> pd.DataFrame: + """Merge fresh power_model with reference v0/naive, write merged tables + per-profile plots.""" + comparison_dir = out_dir / "comparison" + comparison_dir.mkdir(parents=True, exist_ok=True) + + profiles = sorted(fresh["profile"].unique()) + reference = _load_reference_methods(reference_mode_dir, profiles, REUSED_METHODS) + _check_alignment(fresh, reference) + + # v0 comes from the frozen reference run (slow, reference-only, now covering the full 1-12-month grid); + # power_model and the freshly-recomputed naive come from this run. Concatenating both gives the + # three-method comparison across every campaign length the reference covers. + fresh_compare = fresh[fresh["method"].isin(COMPARE_METHODS) & (fresh["method"] != "v0_binned")] + merged = pd.concat([reference, fresh_compare], ignore_index=True) + merged = merged[merged["method"].isin(COMPARE_METHODS)] + + merged.to_csv(comparison_dir / f"merged_results_{mode}.csv", index=False) + for profile in profiles: + prof_rows = merged[merged["profile"] == profile] + summary = leaderboard(prof_rows) + summary.to_csv(comparison_dir / f"leaderboard_{profile}.csv", index=False) + plot_campaign_curves( + summary, + save_path=comparison_dir / f"campaign_curves_{profile}.png", + title=f"{mode} - {profile} (naive vs v0 vs power_model)", + ) + + all_summary = leaderboard(merged) + all_summary.to_csv(comparison_dir / f"leaderboard_all_profiles_{mode}.csv", index=False) + logger.info( + "%s comparison (all profiles):\n%s", + mode, + all_summary[["method", "profile", "campaign_months", "bias", "spread", "score"]].to_string(index=False), + ) + + # Conditional comparison plots for power_model (v0/naive emit no conditional rows — expected). + pm_rows = merged[merged["method"] == "power_model"] + if not pm_rows.empty and "condition" in pm_rows.columns: + cond_lb = conditional_leaderboard(pm_rows) + if not cond_lb.empty: + for profile in profiles: + for condition in cond_lb["condition"].unique(): + subset = _conditional_plot_subset(cond_lb, profile, condition) + if subset.empty: + continue + longest = int(subset["campaign_months"].iloc[0]) + plot_conditional_uplift( + subset, + condition=condition, + save_path=comparison_dir / f"conditional_{profile}_{condition}.png", + title=f"{mode} - {profile} power_model vs {condition} ({longest}mo)", + ) + + logger.info("Wrote %s comparison outputs to %s", mode, comparison_dir) + return merged + + +def main() -> None: + """Run power_model over the overnight cases for each mode, then merge + plot vs v0/naive.""" + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument( + "--reference-dir", + type=Path, + default=_DEFAULT_REFERENCE_DIR, + help="overnight run dir holding prepost/ and toggle/ with the frozen v0 + naive results", + ) + parser.add_argument( + "--output-dir", + type=Path, + default=_DEFAULT_OUTPUT_DIR, + help="where fresh power_model runs and the merged comparison are written", + ) + parser.add_argument( + "--modes", + nargs="+", + choices=["prepost", "toggle"], + default=["prepost", "toggle"], + help="which mode(s) to run (default: both)", + ) + parser.add_argument( + "--profiles", + nargs="+", + choices=sorted(overnight_profiles()), + default=None, + help="restrict to a subset of overnight profiles for fast feedback (default: all seven). " + "e.g. --profiles cp_0pct runs just the placebo. Cannot be combined with --update-baseline.", + ) + parser.add_argument( + "--method-overrides", + type=str, + default=None, + help="JSON dict of PowerModelMethod field overrides for a candidate A/B run, e.g. " + '\'{"matching_vars": ["wind_speed_100m"]}\'. The run is diffed against the benchmark as usual ' + "but no candidate baseline is written (the overridden config is not the committed default).", + ) + parser.add_argument( + "--skip-run", + action="store_true", + help="reuse a previous power_model run under --output-dir; only re-merge and re-plot", + ) + parser.add_argument( + "--baseline-path", + type=Path, + default=_BASELINE_PATH, + help="the committed power_model benchmark JSON to diff against (and to --update-baseline)", + ) + parser.add_argument( + "--update-baseline", + action="store_true", + help="overwrite the recorded benchmark for the run modes with this run (do this deliberately, " + "only when an improvement is accepted); without it, the run is diffed against the benchmark", + ) + parser.add_argument( + "--accept-candidate", + action="store_true", + help="promote the candidate baseline written by the last full sweep (under --output-dir) to the " + "committed --baseline-path and exit — a near-instant accept with no re-run. Then commit the JSON.", + ) + args = parser.parse_args() + if args.profiles is not None and args.update_baseline: + # record_baseline rewrites a mode's cells wholesale, so a subset run would drop the other + # profiles from the committed benchmark. Refuse rather than silently corrupt it. + parser.error("--update-baseline needs the full profile set; do not combine it with --profiles") + method_overrides: dict[str, Any] | None = json.loads(args.method_overrides) if args.method_overrides else None + if method_overrides is not None and args.update_baseline: + # the benchmark freezes the *committed default* config; an overridden run must not be recorded as it + parser.error("--update-baseline records the default config; do not combine it with --method-overrides") + + # Captured before the sweep: it takes hours, so reading HEAD afterwards would stamp the baseline + # with whatever was committed meanwhile rather than the code that actually ran. + git_commit = _git_commit() + _refuse_dirty_update(parser, git_commit=git_commit, update_baseline=args.update_baseline) + + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s", force=True) + reference_dir = args.reference_dir.expanduser() + output_dir = args.output_dir.expanduser() + baseline_path = args.baseline_path.expanduser() + + if args.accept_candidate: + accept_candidate(_candidate_path(output_dir), baseline_path) + return + + lb_by_mode: dict[str, pd.DataFrame] = {} + study_by_mode: dict[str, StudyConfig] = {} + for mode in args.modes: + logger.info("=== %s ===", mode.upper()) + mode_out_dir = output_dir / mode + if args.skip_run: + fresh = _load_fresh_results(mode_out_dir) + if args.profiles is not None: + fresh = fresh[fresh["profile"].isin(args.profiles)] + logger.info("Reusing %d fresh rows from %s", len(fresh), mode_out_dir) + else: + fresh = run_power_model(mode, mode_out_dir, profiles=args.profiles, method_overrides=method_overrides) + if method_overrides is not None: # provenance: which A/B candidate this run was + (mode_out_dir / "method_overrides.json").write_text(json.dumps(method_overrides, indent=2) + "\n") + merge_and_plot(mode, fresh, reference_dir / mode, mode_out_dir) + lb_by_mode[mode] = power_model_leaderboard(fresh) + study_by_mode[mode] = _prepost_study() if mode == "prepost" else _toggle_study() + if not args.update_baseline: + compare_to_benchmark(mode, lb_by_mode[mode], baseline_path, mode_out_dir / "comparison") + conditional_before_after(mode, fresh, baseline_path, mode_out_dir / "comparison") + + if args.update_baseline: + record_baseline(lb_by_mode, study_by_mode, baseline_path, git_commit=git_commit) + elif method_overrides is not None: + logger.info("Overridden run (--method-overrides) — no candidate baseline written (not the default config).") + elif args.profiles is None: + # A full sweep: drop a candidate baseline (seeded from committed) so an accepted improvement is a + # near-instant `--accept-candidate` away, with no second sweep. Subset runs are incomplete -> skip. + candidate = _candidate_path(output_dir) + record_baseline(lb_by_mode, study_by_mode, candidate, seed_path=baseline_path, git_commit=git_commit) + logger.info("Candidate baseline written to %s — accept it with --accept-candidate (no re-run).", candidate) + else: + logger.info("Subset run (--profiles) — no candidate baseline written (it would be incomplete).") + + logger.info("All done. Comparison plots under %s//comparison/", output_dir) + + +if __name__ == "__main__": + main() diff --git a/benchmarking/baselines/study_power_model_compare_baseline.json b/benchmarking/baselines/study_power_model_compare_baseline.json new file mode 100644 index 00000000..4f713008 --- /dev/null +++ b/benchmarking/baselines/study_power_model_compare_baseline.json @@ -0,0 +1,18953 @@ +{ + "schema": "power_model_compare_baseline_v2", + "modes": { + "prepost": { + "recorded_utc": "2026-07-16T10:27:18Z", + "git_commit": "4f07d64", + "n_replicates": 4, + "seed": 0, + "campaign_months": [ + 1, + 2, + 3, + 6, + 12 + ], + "profiles": [ + "cp_0pct", + "cp_minus_10pct", + "cp_plus_10pct", + "cp_plus_3pct", + "rated_plus_5pct", + "ti_dependent_cp", + "ws_dependent_cp" + ], + "cells": [ + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00426993, + "spread": 0.01114444, + "score": 0.01193444 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01403838, + "spread": 0.03027426, + "score": 0.03337076 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00130826, + "spread": 0.02269091, + "score": 0.02272859 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01820391, + "spread": 0.0198524, + "score": 0.02693511 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00342305, + "spread": 0.00288274, + "score": 0.00447521 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.02539078, + "spread": 0.02708964, + "score": 0.03712869 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00552863, + "spread": 0.01140449, + "score": 0.01267391 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.01115517, + "spread": 0.01166995, + "score": 0.0161439 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00898175, + "spread": 0.01381936, + "score": 0.01648171 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00527831, + "spread": 0.0089954, + "score": 0.01042966 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00050982, + "spread": 0.02069985, + "score": 0.02070613 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01241038, + "spread": 0.04168635, + "score": 0.04349447 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.03446985, + "spread": 0.01738451, + "score": 0.03860559 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.34957812, + "spread": 0.65137518, + "score": 0.73925266 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.03689664, + "spread": 0.06119684, + "score": 0.07145919 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.00426993, + "spread": 0.01114444, + "score": 0.01193444 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.02317022, + "spread": 0.09330657, + "score": 0.09614039 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.0015472, + "spread": 0.01819121, + "score": 0.01825688 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00710443, + "spread": 0.01864056, + "score": 0.01994852 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00464541, + "spread": 0.00964761, + "score": 0.01070777 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00674772, + "spread": 0.0116874, + "score": 0.01349544 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.13629254, + "spread": 0.14721031, + "score": 0.20061539 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00543029, + "spread": 0.0128047, + "score": 0.01390857 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00298802, + "spread": 0.01310704, + "score": 0.01344332 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.01380014, + "spread": 0.01770091, + "score": 0.02244473 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -7.415e-05, + "spread": 0.00194037, + "score": 0.00194179 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01487373, + "spread": 0.00897035, + "score": 0.01736937 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00394877, + "spread": 0.00945407, + "score": 0.01024559 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00793958, + "spread": 0.00892941, + "score": 0.01194869 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00187509, + "spread": 0.00202787, + "score": 0.00276192 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00212776, + "spread": 0.00530989, + "score": 0.00572034 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00276168, + "spread": 0.00630778, + "score": 0.00688585 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00227809, + "spread": 0.0, + "score": 0.00227809 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00519397, + "spread": 0.00464831, + "score": 0.00697023 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00170802, + "spread": 0.00259959, + "score": 0.0031105 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00215731, + "spread": 0.00271298, + "score": 0.00346616 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00091056, + "spread": 0.0063966, + "score": 0.00646108 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00163437, + "spread": 0.01443033, + "score": 0.01452259 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00113013, + "spread": 0.03889078, + "score": 0.0389072 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07345375, + "spread": 0.14705128, + "score": 0.16437619 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.52182479, + "spread": 1.16038772, + "score": 1.27232102 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.10537063, + "spread": 0.18412435, + "score": 0.21214322 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.04174168, + "spread": 0.10226973, + "score": 0.11046024 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00245592, + "spread": 0.00957754, + "score": 0.00988741 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00322219, + "spread": 0.00464311, + "score": 0.00565164 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.008385, + "spread": 0.00863245, + "score": 0.01203443 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00156067, + "spread": 0.00270316, + "score": 0.00312134 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.0640303, + "spread": 0.09973422, + "score": 0.11851917 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00648665, + "spread": 0.01111695, + "score": 0.01287102 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00166205, + "spread": 0.00495989, + "score": 0.00523095 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00484893, + "spread": 0.00517295, + "score": 0.00709024 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00206006, + "spread": 0.00188167, + "score": 0.00279008 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00602033, + "spread": 0.00920072, + "score": 0.01099534 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00210015, + "spread": 0.01087959, + "score": 0.01108044 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00739383, + "spread": 0.01100454, + "score": 0.01325777 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00024901, + "spread": 0.00309911, + "score": 0.00310909 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00666666, + "spread": 0.01519318, + "score": 0.01659148 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00557358, + "spread": 0.0103453, + "score": 0.01175117 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00385005, + "spread": 0.00063624, + "score": 0.00390226 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.0068695, + "spread": 0.01246243, + "score": 0.01423033 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00450247, + "spread": 0.00389314, + "score": 0.00595221 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.0006468, + "spread": 0.00126723, + "score": 0.00142275 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.0012469, + "spread": 0.00957785, + "score": 0.00965867 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00521505, + "spread": 0.01738861, + "score": 0.0181538 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.02658984, + "spread": 0.04058415, + "score": 0.048519 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08241847, + "spread": 0.10913265, + "score": 0.13675796 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.20079161, + "spread": 0.39746764, + "score": 0.4453064 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.02701868, + "spread": 0.0517155, + "score": 0.05834811 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.09208905, + "spread": 0.07732789, + "score": 0.12024972 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00221516, + "spread": 0.00702572, + "score": 0.00736666 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00638987, + "spread": 0.00825524, + "score": 0.01043932 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.01123526, + "spread": 0.01156734, + "score": 0.01612558 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.0021629, + "spread": 0.00464052, + "score": 0.00511982 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.0999781, + "spread": 0.10776336, + "score": 0.14699851 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00065364, + "spread": 0.01397082, + "score": 0.0139861 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00198026, + "spread": 0.00946999, + "score": 0.00967482 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00156782, + "spread": 0.00740832, + "score": 0.00757241 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00411231, + "spread": 0.00186471, + "score": 0.00451533 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00733822, + "spread": 0.01301467, + "score": 0.01494092 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00240288, + "spread": 0.00254221, + "score": 0.00349809 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0075062, + "spread": 0.00603545, + "score": 0.00963171 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00049947, + "spread": 0.00181329, + "score": 0.00188082 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00316145, + "spread": 0.00932874, + "score": 0.00984988 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00504652, + "spread": 0.00824801, + "score": 0.00966939 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00458731, + "spread": 0.00216052, + "score": 0.00507063 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00195755, + "spread": 0.00774742, + "score": 0.00799091 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00335936, + "spread": 0.00154154, + "score": 0.00369617 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00209699, + "spread": 0.0018656, + "score": 0.00280675 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00807103, + "spread": 0.00614767, + "score": 0.01014571 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01264494, + "spread": 0.00688185, + "score": 0.01439633 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.01232468, + "spread": 0.01250179, + "score": 0.01755541 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08136692, + "spread": 0.07776325, + "score": 0.11255087 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.19339617, + "spread": 0.12203664, + "score": 0.22868105 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.01706249, + "spread": 0.40158029, + "score": 0.4019426 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.11354135, + "spread": 0.10967945, + "score": 0.15786456 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00157381, + "spread": 0.00322742, + "score": 0.0035907 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.0045376, + "spread": 0.00497337, + "score": 0.00673233 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00614567, + "spread": 0.00436994, + "score": 0.00754093 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00376523, + "spread": 0.0023262, + "score": 0.00442585 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00137305, + "spread": 0.00237819, + "score": 0.0027461 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.11375675, + "spread": 0.06676093, + "score": 0.13190003 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00036526, + "spread": 0.00063265, + "score": 0.00073052 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00942507, + "spread": 0.00772492, + "score": 0.01218632 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.0028878, + "spread": 0.0049096, + "score": 0.00569593 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00408972, + "spread": 0.00636249, + "score": 0.00756354 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00096213, + "spread": 0.00334636, + "score": 0.00348193 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00247886, + "spread": 0.00503641, + "score": 0.00561339 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00079785, + "spread": 0.00642739, + "score": 0.00647672 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00229722, + "spread": 0.00113638, + "score": 0.00256293 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00019342, + "spread": 0.00026237, + "score": 0.00032596 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00094792, + "spread": 0.00718445, + "score": 0.00724672 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00175987, + "spread": 0.0070516, + "score": 0.00726789 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00096213, + "spread": 0.00334636, + "score": 0.00348193 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00245477, + "spread": 0.00332694, + "score": 0.00413454 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -6.33e-06, + "spread": 0.0024154, + "score": 0.00241541 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -2.833e-05, + "spread": 0.004114, + "score": 0.0041141 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00379788, + "spread": 0.00458207, + "score": 0.00595141 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00965358, + "spread": 0.00655176, + "score": 0.01166692 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00101092, + "spread": 0.01234741, + "score": 0.01238872 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.05178133, + "spread": 0.04400123, + "score": 0.06795156 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.37419471, + "spread": 0.39392138, + "score": 0.54331918 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.43925074, + "spread": 0.68039542, + "score": 0.80986366 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.10378407, + "spread": 0.09157735, + "score": 0.13841078 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00102035, + "spread": 0.00284173, + "score": 0.00301936 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00137214, + "spread": 0.00245811, + "score": 0.00281515 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00195442, + "spread": 0.00197181, + "score": 0.00277629 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00050582, + "spread": 0.0007369, + "score": 0.0008938 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00140443, + "spread": 0.00325694, + "score": 0.00354685 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.12418465, + "spread": 0.07005121, + "score": 0.1425798 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00274051, + "spread": 0.01161615, + "score": 0.01193504 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.0013323, + "spread": 0.00721057, + "score": 0.00733262 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00083095, + "spread": 0.00692504, + "score": 0.00697472 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00076612, + "spread": 0.00728604, + "score": 0.00732621 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00384826, + "spread": 0.01034667, + "score": 0.01103914 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01798448, + "spread": 0.0248734, + "score": 0.03069409 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00588722, + "spread": 0.02169078, + "score": 0.02247553 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.02975194, + "spread": 0.02119502, + "score": 0.03652953 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.0007384, + "spread": 0.00331792, + "score": 0.00339909 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.02332791, + "spread": 0.02124848, + "score": 0.03155454 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00700859, + "spread": 0.01282926, + "score": 0.01461883 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.01113027, + "spread": 0.01031433, + "score": 0.01517459 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00733288, + "spread": 0.0109032, + "score": 0.01313966 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00467733, + "spread": 0.00938052, + "score": 0.01048197 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.0011806, + "spread": 0.01880055, + "score": 0.01883758 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01185137, + "spread": 0.03718113, + "score": 0.03902424 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.03172579, + "spread": 0.02103551, + "score": 0.03806598 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.30555752, + "spread": 0.54136676, + "score": 0.6216457 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.00941036, + "spread": 0.11028173, + "score": 0.11068249 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.06506386, + "spread": 0.10149481, + "score": 0.12055912 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.06005579, + "spread": 0.13610279, + "score": 0.1487638 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00367154, + "spread": 0.02047189, + "score": 0.02079852 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.0001943, + "spread": 0.01860817, + "score": 0.01860919 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00495985, + "spread": 0.01078486, + "score": 0.01187069 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00819065, + "spread": 0.01388027, + "score": 0.01611672 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00269719, + "spread": 0.00316539, + "score": 0.00415867 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.0764984, + "spread": 0.07298385, + "score": 0.10572912 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00371969, + "spread": 0.00265996, + "score": 0.00457291 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00775252, + "spread": 0.01259731, + "score": 0.01479168 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00378876, + "spread": 0.01256742, + "score": 0.01312611 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.01055944, + "spread": 0.0165097, + "score": 0.01959776 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -4.35e-06, + "spread": 0.00177896, + "score": 0.00177896 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01480279, + "spread": 0.01052187, + "score": 0.01816128 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00627146, + "spread": 0.01078968, + "score": 0.01247991 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01892891, + "spread": 0.00728206, + "score": 0.02028132 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00734198, + "spread": 0.00315874, + "score": 0.00799264 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00176811, + "spread": 0.00454359, + "score": 0.00487549 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00034386, + "spread": 0.00620007, + "score": 0.0062096 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.02866147, + "spread": 0.0, + "score": 0.02866147 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.01051748, + "spread": 0.00851979, + "score": 0.0135353 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00220454, + "spread": 0.00275592, + "score": 0.00352917 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.001397, + "spread": 0.00367481, + "score": 0.0039314 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00432644, + "spread": 0.00442381, + "score": 0.00618775 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00420609, + "spread": 0.01088413, + "score": 0.01166857 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00539869, + "spread": 0.03605402, + "score": 0.03645598 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07956192, + "spread": 0.13289333, + "score": 0.15488944 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.47777783, + "spread": 1.11158837, + "score": 1.2099175 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.33054658, + "spread": 0.94808219, + "score": 1.00405223 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.08463635, + "spread": 0.11013757, + "score": 0.13890139 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00412392, + "spread": 0.01251363, + "score": 0.01317564 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00281529, + "spread": 0.00329599, + "score": 0.00433467 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00532604, + "spread": 0.00658217, + "score": 0.00846709 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00277639, + "spread": 0.00460643, + "score": 0.00537844 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00163464, + "spread": 0.00278092, + "score": 0.00322577 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.02577533, + "spread": 0.06145086, + "score": 0.06663764 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00380973, + "spread": 0.00256992, + "score": 0.00459549 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.00963368, + "spread": 0.0, + "score": 0.00963368 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00639839, + "spread": 0.01042968, + "score": 0.01223591 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00151072, + "spread": 0.00485983, + "score": 0.00508923 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.0025416, + "spread": 0.00465052, + "score": 0.00529972 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00199417, + "spread": 0.00175822, + "score": 0.00265858 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00625794, + "spread": 0.01457244, + "score": 0.01585932 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.0116314, + "spread": 0.00909188, + "score": 0.01476319 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01927845, + "spread": 0.01282402, + "score": 0.02315414 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00511103, + "spread": 0.00367646, + "score": 0.00629595 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00763161, + "spread": 0.01493865, + "score": 0.01677512 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00378023, + "spread": 0.00905931, + "score": 0.00981637 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.01662755, + "spread": 0.04697228, + "score": 0.04982841 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.01049674, + "spread": 0.0125645, + "score": 0.01637218 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00062672, + "spread": 0.00374288, + "score": 0.00379498 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.0001249, + "spread": 0.00245736, + "score": 0.00246054 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00532275, + "spread": 0.00890481, + "score": 0.01037436 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01014959, + "spread": 0.0156897, + "score": 0.01868638 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.01824445, + "spread": 0.03792589, + "score": 0.04208602 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08918614, + "spread": 0.08933055, + "score": 0.1262304 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.12722958, + "spread": 0.3907416, + "score": 0.41093353 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.08554023, + "spread": 0.26596105, + "score": 0.27937862 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.07155598, + "spread": 0.1889484, + "score": 0.20204394 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00539685, + "spread": 0.01076901, + "score": 0.01204564 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00047273, + "spread": 0.00934848, + "score": 0.00936043 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00720042, + "spread": 0.01102831, + "score": 0.01317079 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00075518, + "spread": 0.00279435, + "score": 0.00289459 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00223408, + "spread": 0.00261857, + "score": 0.0034421 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.05023621, + "spread": 0.0726888, + "score": 0.08835914 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.01000701, + "spread": 0.00900665, + "score": 0.01346329 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.00481684, + "spread": 0.00481684, + "score": 0.00681204 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00039031, + "spread": 0.016853, + "score": 0.01685751 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00243369, + "spread": 0.01013007, + "score": 0.01041831 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00084799, + "spread": 0.00735528, + "score": 0.007404 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.0039361, + "spread": 0.0017386, + "score": 0.00430297 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00690228, + "spread": 0.01429905, + "score": 0.01587779 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.01125114, + "spread": 0.00143332, + "score": 0.01134207 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01843772, + "spread": 0.00634155, + "score": 0.01949781 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00309207, + "spread": 0.0022804, + "score": 0.00384202 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00201187, + "spread": 0.00842594, + "score": 0.0086628 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00325204, + "spread": 0.00616672, + "score": 0.00697167 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.03650845, + "spread": 0.00472694, + "score": 0.03681319 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00086353, + "spread": 0.00727372, + "score": 0.0073248 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00119523, + "spread": 0.00103178, + "score": 0.00157897 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00193604, + "spread": 0.00233197, + "score": 0.0030309 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.01034235, + "spread": 0.00579762, + "score": 0.0118565 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01578415, + "spread": 0.00556343, + "score": 0.01673592 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.01509941, + "spread": 0.0125347, + "score": 0.01962424 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08265843, + "spread": 0.06713304, + "score": 0.10648597 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.04712131, + "spread": 0.35042542, + "score": 0.3535794 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.11349187, + "spread": 0.36783953, + "score": 0.38494977 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.03984096, + "spread": 0.1732238, + "score": 0.17774641 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00459812, + "spread": 0.00425721, + "score": 0.0062663 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.0009016, + "spread": 0.00511846, + "score": 0.00519726 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00337356, + "spread": 0.00411458, + "score": 0.00532078 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00192021, + "spread": 0.00048286, + "score": 0.00197999 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00253225, + "spread": 0.00243456, + "score": 0.00351275 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.05547811, + "spread": 0.03051691, + "score": 0.06331747 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00597601, + "spread": 0.00519317, + "score": 0.00791718 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.01285704, + "spread": 0.00302876, + "score": 0.01320897 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.008709, + "spread": 0.00788457, + "score": 0.0117479 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00265028, + "spread": 0.00498366, + "score": 0.00564455 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00535685, + "spread": 0.00566817, + "score": 0.00779897 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00097627, + "spread": 0.00315056, + "score": 0.00329836 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00146815, + "spread": 0.00548167, + "score": 0.00567488 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00838365, + "spread": 0.00510736, + "score": 0.00981686 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01161496, + "spread": 0.00258368, + "score": 0.01189886 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00316989, + "spread": 0.00024445, + "score": 0.0031793 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00033869, + "spread": 0.00652522, + "score": 0.006534 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00082315, + "spread": 0.00668295, + "score": 0.00673346 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.00779326, + "spread": 0.05332963, + "score": 0.05389605 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00106047, + "spread": 0.0049238, + "score": 0.00503671 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00073637, + "spread": 0.00227546, + "score": 0.00239165 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00065292, + "spread": 0.00401113, + "score": 0.00406392 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00582588, + "spread": 0.00395896, + "score": 0.00704373 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01230101, + "spread": 0.00554853, + "score": 0.01349448 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00702036, + "spread": 0.01124452, + "score": 0.01325612 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.05487793, + "spread": 0.04017211, + "score": 0.06801019 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.34051897, + "spread": 0.32206093, + "score": 0.4686965 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.08474655, + "spread": 0.98931339, + "score": 0.99293654 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.08501932, + "spread": 0.08819053, + "score": 0.12249838 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00452846, + "spread": 0.00219194, + "score": 0.00503106 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.002707, + "spread": 0.00293234, + "score": 0.0039908 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00013067, + "spread": 0.0018325, + "score": 0.00183716 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00064964, + "spread": 0.00155113, + "score": 0.00168167 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00166555, + "spread": 0.00518857, + "score": 0.00544934 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.06933752, + "spread": 0.03117433, + "score": 0.07602323 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00624964, + "spread": 0.01389321, + "score": 0.01523415 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.00832078, + "spread": 0.00396427, + "score": 0.00921688 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.00421786, + "spread": 0.00596496, + "score": 0.00730556 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00133231, + "spread": 0.00734114, + "score": 0.00746106 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 4.406e-05, + "spread": 0.00683774, + "score": 0.00683789 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00197835, + "spread": 0.00678884, + "score": 0.00707123 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00469112, + "spread": 0.0119443, + "score": 0.01283249 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.02252384, + "spread": 0.03645112, + "score": 0.04284866 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00250423, + "spread": 0.02653637, + "score": 0.02665427 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01282654, + "spread": 0.02217634, + "score": 0.02561855 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.0040571, + "spread": 0.004613, + "score": 0.00614328 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.02797461, + "spread": 0.03122669, + "score": 0.04192476 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.01081487, + "spread": 0.00933839, + "score": 0.0142887 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.0118305, + "spread": 0.02754161, + "score": 0.02997501 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.01047774, + "spread": 0.01762825, + "score": 0.02050703 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00547375, + "spread": 0.00902041, + "score": 0.01055129 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00063771, + "spread": 0.02136928, + "score": 0.02137879 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01044814, + "spread": 0.04400134, + "score": 0.04522479 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.02645185, + "spread": 0.01221653, + "score": 0.02913664 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.37481409, + "spread": 0.74580417, + "score": 0.83469124 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.05526605, + "spread": 0.06943106, + "score": 0.08874124 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.07360324, + "spread": 0.10706847, + "score": 0.12992727 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.06058553, + "spread": 0.08668914, + "score": 0.10576207 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.0001018, + "spread": 0.01704679, + "score": 0.01704709 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.01415104, + "spread": 0.0202714, + "score": 0.02472209 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00454642, + "spread": 0.00952511, + "score": 0.01055451 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00593339, + "spread": 0.0105849, + "score": 0.01213446 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00269719, + "spread": 0.00316539, + "score": 0.00415867 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.2254042, + "spread": 0.22408564, + "score": 0.31783868 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00371969, + "spread": 0.00265996, + "score": 0.00457291 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00740578, + "spread": 0.01663438, + "score": 0.01820847 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00293746, + "spread": 0.01317416, + "score": 0.01349767 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.01674645, + "spread": 0.01850764, + "score": 0.02495949 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00014375, + "spread": 0.00210202, + "score": 0.00210693 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.02452249, + "spread": 0.02011447, + "score": 0.03171663 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.01124838, + "spread": 0.00995243, + "score": 0.01501922 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00494601, + "spread": 0.00792746, + "score": 0.00934386 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00134241, + "spread": 0.00468343, + "score": 0.00487202 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.0078345, + "spread": 0.0039501, + "score": 0.00877398 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00055975, + "spread": 0.00485872, + "score": 0.00489086 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.02410822, + "spread": 0.0, + "score": 0.02410822 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00112255, + "spread": 0.00539816, + "score": 0.00551364 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.0059868, + "spread": 0.00492861, + "score": 0.00775455 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00273603, + "spread": 0.00184004, + "score": 0.00329721 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00254849, + "spread": 0.00847252, + "score": 0.0088475 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00869233, + "spread": 0.01618735, + "score": 0.01837353 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.01281866, + "spread": 0.03652679, + "score": 0.03871078 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.05697343, + "spread": 0.17676046, + "score": 0.18571546 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.55560351, + "spread": 1.2501431, + "score": 1.36804716 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.58740607, + "spread": 1.38642431, + "score": 1.50572848 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.05952777, + "spread": 0.17237228, + "score": 0.18236162 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00081554, + "spread": 0.00504111, + "score": 0.00510666 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00869986, + "spread": 0.00354601, + "score": 0.00939478 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.01174806, + "spread": 0.00978863, + "score": 0.01529164 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00174157, + "spread": 0.00321896, + "score": 0.00365989 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00163464, + "spread": 0.00278092, + "score": 0.00322577 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.1143688, + "spread": 0.12582742, + "score": 0.17003753 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00380973, + "spread": 0.00256992, + "score": 0.00459549 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00963368, + "spread": 0.0, + "score": 0.00963368 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.01070271, + "spread": 0.01629079, + "score": 0.01949199 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00096938, + "spread": 0.00260803, + "score": 0.00278235 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00705426, + "spread": 0.00539844, + "score": 0.00888289 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00212595, + "spread": 0.00200591, + "score": 0.00292289 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00761734, + "spread": 0.01347733, + "score": 0.01548103 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00336395, + "spread": 0.01308516, + "score": 0.01351065 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00226052, + "spread": 0.0110063, + "score": 0.01123604 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00146029, + "spread": 0.00331361, + "score": 0.00362111 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00942877, + "spread": 0.01750438, + "score": 0.01988228 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00427219, + "spread": 0.00912289, + "score": 0.01007366 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.02432763, + "spread": 0.04824474, + "score": 0.05403136 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00678368, + "spread": 0.01397751, + "score": 0.0155367 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00792632, + "spread": 0.00474794, + "score": 0.00923956 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00163904, + "spread": 0.0009257, + "score": 0.00188238 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00151043, + "spread": 0.01074864, + "score": 0.01085424 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00041882, + "spread": 0.02340316, + "score": 0.0234069 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.03139314, + "spread": 0.04830329, + "score": 0.05760848 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08704528, + "spread": 0.12343039, + "score": 0.15103623 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.26477838, + "spread": 0.41353733, + "score": 0.49104044 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.02631239, + "spread": 0.34255718, + "score": 0.34356624 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.15160466, + "spread": 0.19594569, + "score": 0.24774722 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.0009676, + "spread": 0.00562125, + "score": 0.00570392 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.01088499, + "spread": 0.00973051, + "score": 0.0146002 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.01202918, + "spread": 0.01393985, + "score": 0.01841251 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00196339, + "spread": 0.00697015, + "score": 0.00724141 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00223408, + "spread": 0.00261857, + "score": 0.0034421 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.13931809, + "spread": 0.15706694, + "score": 0.20995131 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.01000701, + "spread": 0.00900665, + "score": 0.01346329 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00481684, + "spread": 0.00481684, + "score": 0.00681204 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.0015844, + "spread": 0.01735778, + "score": 0.01742994 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00178431, + "spread": 0.00855707, + "score": 0.00874112 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00319845, + "spread": 0.00694423, + "score": 0.00764542 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00429022, + "spread": 0.00199034, + "score": 0.00472943 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00328689, + "spread": 0.01603097, + "score": 0.01636447 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00479896, + "spread": 0.00583736, + "score": 0.00755677 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00367001, + "spread": 0.00704511, + "score": 0.00794372 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00214068, + "spread": 0.00213288, + "score": 0.00302187 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00033009, + "spread": 0.00977297, + "score": 0.00977854 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00480699, + "spread": 0.00803402, + "score": 0.00936229 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.02733043, + "spread": 0.00040764, + "score": 0.02733347 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00255402, + "spread": 0.01005698, + "score": 0.01037621 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00553739, + "spread": 0.00253473, + "score": 0.00608995 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00212832, + "spread": 0.00171646, + "score": 0.00273422 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00529421, + "spread": 0.00708544, + "score": 0.00884489 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00999186, + "spread": 0.00875742, + "score": 0.01328644 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00995044, + "spread": 0.01508466, + "score": 0.01807092 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07519873, + "spread": 0.08392378, + "score": 0.11268562 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.17891073, + "spread": 0.14836137, + "score": 0.23242234 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.02649062, + "spread": 0.50542914, + "score": 0.50612288 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.1114352, + "spread": 0.15069909, + "score": 0.1874247 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00139034, + "spread": 0.00330379, + "score": 0.00358442 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.0092899, + "spread": 0.00407708, + "score": 0.01014519 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00794191, + "spread": 0.004687, + "score": 0.00922182 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00428739, + "spread": 0.00396775, + "score": 0.00584164 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -9.463e-05, + "spread": 0.00343065, + "score": 0.00343196 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.17605222, + "spread": 0.11858626, + "score": 0.21226654 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00512954, + "spread": 0.00534205, + "score": 0.00740605 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.01285704, + "spread": 0.00302876, + "score": 0.01320897 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00985822, + "spread": 0.00956423, + "score": 0.01373532 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00374436, + "spread": 0.00423361, + "score": 0.00565187 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00267045, + "spread": 0.00663471, + "score": 0.00715197 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00094456, + "spread": 0.00354445, + "score": 0.00366815 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00450818, + "spread": 0.004753, + "score": 0.00655093 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00494676, + "spread": 0.00821588, + "score": 0.00959015 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00017879, + "spread": 0.00120204, + "score": 0.00121527 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.0006107, + "spread": 0.00038773, + "score": 0.00072338 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00157557, + "spread": 0.00744062, + "score": 0.0076056 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00026171, + "spread": 0.00736157, + "score": 0.00736622 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00971409, + "spread": 0.04841127, + "score": 0.04937625 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00670118, + "spread": 0.00167178, + "score": 0.00690657 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00065821, + "spread": 0.00255107, + "score": 0.00263461 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00045186, + "spread": 0.0042961, + "score": 0.0043198 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00204379, + "spread": 0.00528983, + "score": 0.00567093 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00714932, + "spread": 0.00759208, + "score": 0.01042844 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00302996, + "spread": 0.01528507, + "score": 0.01558249 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.05127752, + "spread": 0.0523248, + "score": 0.07326165 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.41272246, + "spread": 0.46218289, + "score": 0.61963929 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.78581652, + "spread": 0.81202171, + "score": 1.12999419 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.12688772, + "spread": 0.11616809, + "score": 0.17203348 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00265643, + "spread": 0.00414965, + "score": 0.00492709 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00525397, + "spread": 0.00233223, + "score": 0.00574835 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00326977, + "spread": 0.00239038, + "score": 0.00405034 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00114216, + "spread": 0.00096137, + "score": 0.0014929 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00048834, + "spread": 0.00178031, + "score": 0.00184607 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.17628527, + "spread": 0.11198741, + "score": 0.20884845 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00195877, + "spread": 0.01052302, + "score": 0.01070377 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00832078, + "spread": 0.00396427, + "score": 0.00921688 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.00421786, + "spread": 0.00596496, + "score": 0.00730556 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00211588, + "spread": 0.00779494, + "score": 0.008077 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00126096, + "spread": 0.00711571, + "score": 0.00722657 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00036179, + "spread": 0.00767475, + "score": 0.00768327 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00439269, + "spread": 0.01138657, + "score": 0.01220449 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01871322, + "spread": 0.03402892, + "score": 0.03883494 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00110039, + "spread": 0.02579568, + "score": 0.02581914 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01603954, + "spread": 0.02220126, + "score": 0.0273891 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00423264, + "spread": 0.00330017, + "score": 0.00536715 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.0270099, + "spread": 0.02728048, + "score": 0.03838957 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00886402, + "spread": 0.01041873, + "score": 0.01367921 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.01190109, + "spread": 0.01641543, + "score": 0.02027566 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00949785, + "spread": 0.01497328, + "score": 0.01773157 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00504978, + "spread": 0.00906777, + "score": 0.01037905 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00022294, + "spread": 0.02047476, + "score": 0.02047597 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01062733, + "spread": 0.04100332, + "score": 0.04235814 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.02997567, + "spread": 0.01458315, + "score": 0.03333481 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.35172582, + "spread": 0.68430158, + "score": 0.76940218 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.04421406, + "spread": 0.05556885, + "score": 0.07101254 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.02506632, + "spread": 0.03559409, + "score": 0.04353458 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.01092867, + "spread": 0.09237246, + "score": 0.0930167 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00244911, + "spread": 0.01846087, + "score": 0.01862261 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.01046245, + "spread": 0.02072387, + "score": 0.02321511 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00439248, + "spread": 0.0102322, + "score": 0.01113516 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00672058, + "spread": 0.01173258, + "score": 0.01352108 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00080916, + "spread": 0.00094962, + "score": 0.0012476 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.16921406, + "spread": 0.16596524, + "score": 0.23701869 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00111591, + "spread": 0.00079799, + "score": 0.00137187 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00856047, + "spread": 0.01611924, + "score": 0.01825134 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00379671, + "spread": 0.01330774, + "score": 0.01383874 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.01503328, + "spread": 0.01800518, + "score": 0.02345605 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -9.481e-05, + "spread": 0.00198906, + "score": 0.00199132 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.0223914, + "spread": 0.01363076, + "score": 0.02621397 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00615587, + "spread": 0.01029231, + "score": 0.01199276 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0065776, + "spread": 0.00795394, + "score": 0.01032133 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00150887, + "spread": 0.00284859, + "score": 0.00322353 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00349704, + "spread": 0.00391017, + "score": 0.00524583 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00352724, + "spread": 0.0040454, + "score": 0.00536719 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.00563693, + "spread": 0.0, + "score": 0.00563693 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00403418, + "spread": 0.00482131, + "score": 0.00628646 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00299665, + "spread": 0.00280111, + "score": 0.00410196 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00223153, + "spread": 0.00257462, + "score": 0.00340711 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -7.599e-05, + "spread": 0.00644377, + "score": 0.00644422 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00438893, + "spread": 0.01486293, + "score": 0.0154974 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00741036, + "spread": 0.03565876, + "score": 0.03642061 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.0681238, + "spread": 0.15929001, + "score": 0.17324595 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.53615497, + "spread": 1.20532024, + "score": 1.31918878 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.24808964, + "spread": 0.54010193, + "score": 0.59435559 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.03227703, + "spread": 0.13394449, + "score": 0.13777857 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00223178, + "spread": 0.00847151, + "score": 0.00876055 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00448531, + "spread": 0.00422648, + "score": 0.00616288 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00903769, + "spread": 0.00864944, + "score": 0.0125097 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00175427, + "spread": 0.00309921, + "score": 0.00356126 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00049039, + "spread": 0.00083428, + "score": 0.00096773 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.08268063, + "spread": 0.10667318, + "score": 0.1349639 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00114292, + "spread": 0.00077098, + "score": 0.00137865 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.0028901, + "spread": 0.0, + "score": 0.0028901 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.0097422, + "spread": 0.01365506, + "score": 0.01677413 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00149624, + "spread": 0.0024431, + "score": 0.00286487 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00508158, + "spread": 0.00542516, + "score": 0.00743336 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00207983, + "spread": 0.00191886, + "score": 0.00282979 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00875119, + "spread": 0.01322455, + "score": 0.01585787 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00030999, + "spread": 0.01159339, + "score": 0.01159754 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00611598, + "spread": 0.01153775, + "score": 0.01305852 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00032244, + "spread": 0.00318224, + "score": 0.00319853 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00782541, + "spread": 0.01506843, + "score": 0.01697924 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00534537, + "spread": 0.00969039, + "score": 0.01106692 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00999333, + "spread": 0.01491879, + "score": 0.01795653 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00743207, + "spread": 0.01301775, + "score": 0.01498992 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00527989, + "spread": 0.00412647, + "score": 0.00670111 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00074087, + "spread": 0.00077018, + "score": 0.00106868 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00048262, + "spread": 0.00986616, + "score": 0.00987796 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00382774, + "spread": 0.01998095, + "score": 0.02034429 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.02769888, + "spread": 0.04748423, + "score": 0.05497254 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08332139, + "spread": 0.10152118, + "score": 0.13133547 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.21697915, + "spread": 0.38680353, + "score": 0.44350527 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.01637192, + "spread": 0.14071456, + "score": 0.14166378 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.14686144, + "spread": 0.19041793, + "score": 0.24047301 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00152228, + "spread": 0.00671371, + "score": 0.00688413 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00764496, + "spread": 0.00905634, + "score": 0.0118517 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.01081997, + "spread": 0.0122196, + "score": 0.01632147 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.0017901, + "spread": 0.00491269, + "score": 0.00522867 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00067022, + "spread": 0.00078557, + "score": 0.00103263 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.1067554, + "spread": 0.1296244, + "score": 0.16792618 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.0030021, + "spread": 0.002702, + "score": 0.00403899 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00144505, + "spread": 0.00144505, + "score": 0.00204361 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00069284, + "spread": 0.01593002, + "score": 0.01594508 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00267275, + "spread": 0.0083816, + "score": 0.00879743 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00180599, + "spread": 0.00688135, + "score": 0.00711439 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00416586, + "spread": 0.00190254, + "score": 0.00457975 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00463286, + "spread": 0.01513756, + "score": 0.01583064 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00059253, + "spread": 0.00321396, + "score": 0.00326813 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00548146, + "spread": 0.00602771, + "score": 0.00814738 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00099907, + "spread": 0.00168242, + "score": 0.0019567 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00251916, + "spread": 0.00922088, + "score": 0.00955881 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00590752, + "spread": 0.0079081, + "score": 0.00987101 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.00498765, + "spread": 0.00139027, + "score": 0.00517779 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00152451, + "spread": 0.00906667, + "score": 0.00919395 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00415223, + "spread": 0.00166808, + "score": 0.00447477 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00210052, + "spread": 0.00192212, + "score": 0.00284723 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00685613, + "spread": 0.00653401, + "score": 0.009471 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01172617, + "spread": 0.00824519, + "score": 0.01433479 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.01161678, + "spread": 0.01389679, + "score": 0.01811272 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07594206, + "spread": 0.07663778, + "score": 0.10789136 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.23474407, + "spread": 0.06581682, + "score": 0.24379629 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.0718369, + "spread": 0.38219817, + "score": 0.3888907 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.08564754, + "spread": 0.14083608, + "score": 0.16483417 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00069406, + "spread": 0.00336236, + "score": 0.00343325 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00594013, + "spread": 0.00469701, + "score": 0.00757279 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00673786, + "spread": 0.00434097, + "score": 0.00801516 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00367086, + "spread": 0.00268145, + "score": 0.00454592 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00076609, + "spread": 0.00229501, + "score": 0.0024195 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.13525274, + "spread": 0.08355387, + "score": 0.15897973 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00162782, + "spread": 0.00157318, + "score": 0.00226378 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00385711, + "spread": 0.00090863, + "score": 0.00396269 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00868033, + "spread": 0.00829359, + "score": 0.01200549 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00340421, + "spread": 0.00463928, + "score": 0.00575426 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00404146, + "spread": 0.00641582, + "score": 0.00758263 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00095758, + "spread": 0.00340573, + "score": 0.00353779 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00293732, + "spread": 0.004981, + "score": 0.00578258 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00103005, + "spread": 0.00648312, + "score": 0.00656444 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00088873, + "spread": 0.00084983, + "score": 0.00122965 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00016494, + "spread": 0.00034275, + "score": 0.00038037 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00018383, + "spread": 0.00718337, + "score": 0.00718572 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00177345, + "spread": 0.00756737, + "score": 0.0077724 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00358844, + "spread": 0.01298148, + "score": 0.01346832 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00391575, + "spread": 0.00223868, + "score": 0.00451052 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00010413, + "spread": 0.00245938, + "score": 0.00246158 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00017205, + "spread": 0.00418455, + "score": 0.00418808 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00331532, + "spread": 0.00494331, + "score": 0.00595212 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00893412, + "spread": 0.00686222, + "score": 0.01126537 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00056676, + "spread": 0.01321297, + "score": 0.01322511 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.0530807, + "spread": 0.04638849, + "score": 0.07049434 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.38941328, + "spread": 0.4195148, + "score": 0.57239442 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.5425848, + "spread": 0.67031901, + "score": 0.8623954 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.11606242, + "spread": 0.09257672, + "score": 0.1484619 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 3.631e-05, + "spread": 0.00314395, + "score": 0.00314416 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00246796, + "spread": 0.00235494, + "score": 0.00341124 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00224025, + "spread": 0.00208526, + "score": 0.00306056 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00060449, + "spread": 0.00086419, + "score": 0.00105463 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00109056, + "spread": 0.00265989, + "score": 0.00287477 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.13885579, + "spread": 0.08159688, + "score": 0.16105584 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0009294, + "spread": 0.01197101, + "score": 0.01200703 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00249623, + "spread": 0.00118928, + "score": 0.00276506 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.00126536, + "spread": 0.00178949, + "score": 0.00219167 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00128968, + "spread": 0.00758608, + "score": 0.00769493 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00083683, + "spread": 0.00699539, + "score": 0.00704526 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.0005884, + "spread": 0.0074146, + "score": 0.00743791 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00429428, + "spread": 0.01132645, + "score": 0.01211319 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01471574, + "spread": 0.0281923, + "score": 0.03180187 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00434809, + "spread": 0.02635468, + "score": 0.02671095 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.02553907, + "spread": 0.02618097, + "score": 0.03657441 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.01023427, + "spread": 0.02179657, + "score": 0.02407967 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.02372059, + "spread": 0.02537939, + "score": 0.03473874 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00746727, + "spread": 0.00972005, + "score": 0.01225722 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.01166778, + "spread": 0.00760655, + "score": 0.01392827 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00883065, + "spread": 0.01299074, + "score": 0.01570795 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00533621, + "spread": 0.00988153, + "score": 0.01123031 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00137974, + "spread": 0.02048129, + "score": 0.02052771 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01268986, + "spread": 0.04054447, + "score": 0.04248396 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.03172131, + "spread": 0.01799019, + "score": 0.03646763 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.35107716, + "spread": 0.66333932, + "score": 0.75051597 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.04768047, + "spread": 0.05578494, + "score": 0.0733852 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.00947451, + "spread": 0.01380549, + "score": 0.01674389 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00306247, + "spread": 0.08018243, + "score": 0.08024089 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.0049816, + "spread": 0.02014672, + "score": 0.02075347 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00620599, + "spread": 0.0199424, + "score": 0.02088572 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.0192129, + "spread": 0.03165775, + "score": 0.03703173 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.02905401, + "spread": 0.03554335, + "score": 0.04590714 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.04836857, + "spread": 0.00159402, + "score": 0.04839482 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.13798073, + "spread": 0.14622772, + "score": 0.20105031 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.04807698, + "spread": 0.00139316, + "score": 0.04809716 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00769592, + "spread": 0.01170343, + "score": 0.01400705 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00145086, + "spread": 0.01310795, + "score": 0.013188 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.01071466, + "spread": 0.01720786, + "score": 0.02027102 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -6.683e-05, + "spread": 0.00197798, + "score": 0.00197911 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01775934, + "spread": 0.01081611, + "score": 0.02079381 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00040418, + "spread": 0.01076934, + "score": 0.01077692 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01246173, + "spread": 0.00882836, + "score": 0.01527202 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00346831, + "spread": 0.00227903, + "score": 0.00415008 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00128, + "spread": 0.00628149, + "score": 0.00641058 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00279179, + "spread": 0.00579763, + "score": 0.0064348 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.01573493, + "spread": 0.0, + "score": 0.01573493 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00860291, + "spread": 0.0081331, + "score": 0.01183881 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00019069, + "spread": 0.00213692, + "score": 0.00214542 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00172563, + "spread": 0.00341239, + "score": 0.00382389 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00216929, + "spread": 0.00535878, + "score": 0.0057812 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00031865, + "spread": 0.01420455, + "score": 0.01420812 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00109011, + "spread": 0.03863048, + "score": 0.03864586 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07244372, + "spread": 0.14583859, + "score": 0.16284037 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.48443718, + "spread": 1.08792829, + "score": 1.1909103 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.09159499, + "spread": 0.18544505, + "score": 0.20683208 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.0818014, + "spread": 0.1102106, + "score": 0.13725103 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00537257, + "spread": 0.01077628, + "score": 0.01204129 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00264744, + "spread": 0.00536335, + "score": 0.00598118 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00948326, + "spread": 0.00918326, + "score": 0.01320092 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.03444503, + "spread": 0.02608581, + "score": 0.04320798 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.04878756, + "spread": 0.00132291, + "score": 0.04880549 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.06480771, + "spread": 0.10600892, + "score": 0.12424947 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.04792307, + "spread": 0.00123925, + "score": 0.04793909 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.04487183, + "spread": 0.0, + "score": 0.04487183 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.04956386, + "spread": 0.0, + "score": 0.04956386 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00588838, + "spread": 0.01173422, + "score": 0.01312878 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00284963, + "spread": 0.00395535, + "score": 0.00487495 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00236987, + "spread": 0.0052368, + "score": 0.00574807 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00210247, + "spread": 0.00191523, + "score": 0.00284403 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00768078, + "spread": 0.01011721, + "score": 0.01270245 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.0068387, + "spread": 0.01087394, + "score": 0.01284564 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01209698, + "spread": 0.01296628, + "score": 0.01773306 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00202726, + "spread": 0.00350005, + "score": 0.00404476 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00655081, + "spread": 0.01522438, + "score": 0.01657392 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00516104, + "spread": 0.00933226, + "score": 0.0106643 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.01887459, + "spread": 0.00182764, + "score": 0.01896287 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.0087472, + "spread": 0.01415733, + "score": 0.01664162 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00260534, + "spread": 0.00387762, + "score": 0.00467159 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00030891, + "spread": 0.00173394, + "score": 0.00176124 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00365484, + "spread": 0.00952955, + "score": 0.01020638 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00924582, + "spread": 0.01768661, + "score": 0.01995748 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.02381222, + "spread": 0.0403035, + "score": 0.04681232 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08787257, + "spread": 0.11065362, + "score": 0.14130043 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.19576611, + "spread": 0.37626555, + "score": 0.42414636 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.01562162, + "spread": 0.05983303, + "score": 0.06183872 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.05790435, + "spread": 0.14894849, + "score": 0.15980791 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.0052374, + "spread": 0.00956085, + "score": 0.01090138 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00439853, + "spread": 0.008882, + "score": 0.00991146 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.01105979, + "spread": 0.01141505, + "score": 0.0158941 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.0231453, + "spread": 0.02652219, + "score": 0.0352013 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.04853016, + "spread": 0.001238, + "score": 0.04854595 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.09577589, + "spread": 0.10841197, + "score": 0.14465882 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.04492176, + "spread": 0.00440256, + "score": 0.04513698 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.04720487, + "spread": 0.00233305, + "score": 0.04726249 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.04956386, + "spread": 0.0, + "score": 0.04956386 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.0029589, + "spread": 0.01682361, + "score": 0.01708183 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00094612, + "spread": 0.00976476, + "score": 0.00981049 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.0010468, + "spread": 0.00734971, + "score": 0.00742388 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00419323, + "spread": 0.00188561, + "score": 0.00459769 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00705461, + "spread": 0.01536865, + "score": 0.01691044 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.0065577, + "spread": 0.00186177, + "score": 0.00681686 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01040807, + "spread": 0.0070964, + "score": 0.0125971 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00077275, + "spread": 0.00194535, + "score": 0.00209321 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00416173, + "spread": 0.00918967, + "score": 0.01008811 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00580073, + "spread": 0.00768208, + "score": 0.00962616 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.0197497, + "spread": 0.00243129, + "score": 0.01989879 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00039165, + "spread": 0.00756584, + "score": 0.00757597 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.0024974, + "spread": 0.00122736, + "score": 0.0027827 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00206803, + "spread": 0.00224207, + "score": 0.00305019 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00977683, + "spread": 0.00605569, + "score": 0.01150034 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01473194, + "spread": 0.00675888, + "score": 0.01620841 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.01361913, + "spread": 0.01496244, + "score": 0.02023253 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08050641, + "spread": 0.07506602, + "score": 0.11007356 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.19710426, + "spread": 0.11207981, + "score": 0.22674209 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.02683948, + "spread": 0.36925156, + "score": 0.3702257 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.08627481, + "spread": 0.17586774, + "score": 0.19588978 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00387624, + "spread": 0.00393354, + "score": 0.0055225 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00237809, + "spread": 0.00562188, + "score": 0.00610416 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00554474, + "spread": 0.00463738, + "score": 0.00722837 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00290452, + "spread": 0.00142409, + "score": 0.00323485 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.03526341, + "spread": 0.02342437, + "score": 0.04233449 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.11452406, + "spread": 0.0698213, + "score": 0.13412969 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.03435083, + "spread": 0.02181242, + "score": 0.04069105 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.04333881, + "spread": 0.00157149, + "score": 0.04336729 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.02478194, + "spread": 0.02478192, + "score": 0.03504694 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00964312, + "spread": 0.00865435, + "score": 0.01295714 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.0035372, + "spread": 0.00524403, + "score": 0.00632548 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00531529, + "spread": 0.00642221, + "score": 0.00833649 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00099684, + "spread": 0.00341025, + "score": 0.00355295 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00061921, + "spread": 0.00510911, + "score": 0.0051465 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00390236, + "spread": 0.0062922, + "score": 0.00740407 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00466426, + "spread": 0.00119937, + "score": 0.00481599 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00116816, + "spread": 0.00039379, + "score": 0.00123275 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00139308, + "spread": 0.00727222, + "score": 0.00740445 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00272843, + "spread": 0.0070663, + "score": 0.00757475 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.02100137, + "spread": 0.00269335, + "score": 0.02117337 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00087444, + "spread": 0.00443805, + "score": 0.00452338 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00045335, + "spread": 0.00256036, + "score": 0.00260019 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00031697, + "spread": 0.00419068, + "score": 0.00420265 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00508659, + "spread": 0.00457293, + "score": 0.00683996 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.0115283, + "spread": 0.00640595, + "score": 0.01318855 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00319144, + "spread": 0.01276425, + "score": 0.01315717 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.05529595, + "spread": 0.04671727, + "score": 0.07238885 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.37350333, + "spread": 0.38555721, + "score": 0.53680453 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.44598939, + "spread": 0.68459115, + "score": 0.81705053 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.11002322, + "spread": 0.10217321, + "score": 0.15014817 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00301787, + "spread": 0.00238611, + "score": 0.00384722 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00042106, + "spread": 0.00275645, + "score": 0.00278842 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00102344, + "spread": 0.00196839, + "score": 0.00221855 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00017397, + "spread": 0.00125644, + "score": 0.00126843 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0014506, + "spread": 0.00423985, + "score": 0.00448113 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.12426875, + "spread": 0.07054484, + "score": 0.1428961 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.0080647, + "spread": 0.02570926, + "score": 0.02694449 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.04572629, + "spread": 0.00201169, + "score": 0.04577052 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.0312851, + "spread": 0.02228091, + "score": 0.03840829 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00172505, + "spread": 0.00786196, + "score": 0.00804898 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00070242, + "spread": 0.00733085, + "score": 0.00736443 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00164808, + "spread": 0.00751702, + "score": 0.00769556 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00449064, + "spread": 0.01159994, + "score": 0.01243883 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.0133029, + "spread": 0.03217985, + "score": 0.03482111 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00154385, + "spread": 0.02629813, + "score": 0.02634341 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01201269, + "spread": 0.02166848, + "score": 0.02477555 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00419692, + "spread": 0.00421738, + "score": 0.00594983 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.02766271, + "spread": 0.02926485, + "score": 0.0402698 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00698286, + "spread": 0.00856445, + "score": 0.01105035 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.02928841, + "spread": 0.03298634, + "score": 0.04411247 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.01483987, + "spread": 0.01831952, + "score": 0.02357597 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00614511, + "spread": 0.00938917, + "score": 0.01122136 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00590934, + "spread": 0.02141461, + "score": 0.02221499 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.0243693, + "spread": 0.04178972, + "score": 0.04837606 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.05301411, + "spread": 0.01568386, + "score": 0.05528544 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.37443452, + "spread": 0.6671997, + "score": 0.76508604 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.07572818, + "spread": 0.0509734, + "score": 0.09128551 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.03926698, + "spread": 0.01300946, + "score": 0.04136595 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.041602, + "spread": 0.08268889, + "score": 0.09256446 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00018104, + "spread": 0.01674569, + "score": 0.01674667 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.01060232, + "spread": 0.01970981, + "score": 0.02238048 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00449421, + "spread": 0.01026217, + "score": 0.01120313 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00654954, + "spread": 0.01149304, + "score": 0.01322824 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00101017, + "spread": 0.00110969, + "score": 0.00150062 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.15918519, + "spread": 0.17707135, + "score": 0.23810541 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00146186, + "spread": 0.00076184, + "score": 0.00164846 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00460095, + "spread": 0.01428279, + "score": 0.01500556 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00243233, + "spread": 0.01274009, + "score": 0.0129702 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.01547368, + "spread": 0.01800735, + "score": 0.02374235 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -9.79e-05, + "spread": 0.00204292, + "score": 0.00204526 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01901503, + "spread": 0.01177995, + "score": 0.02236825 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00948094, + "spread": 0.00930557, + "score": 0.01328465 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00355395, + "spread": 0.00713565, + "score": 0.0079717 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00169377, + "spread": 0.00404068, + "score": 0.00438132 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00194248, + "spread": 0.00486307, + "score": 0.00523666 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00344207, + "spread": 0.00305898, + "score": 0.00460491 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.0471605, + "spread": 0.0, + "score": 0.0471605 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.0097794, + "spread": 0.01469675, + "score": 0.01765307 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00115104, + "spread": 0.00545578, + "score": 0.00557588 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00336632, + "spread": 0.0032976, + "score": 0.00471236 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00445479, + "spread": 0.00863603, + "score": 0.00971731 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01038448, + "spread": 0.01294471, + "score": 0.01659527 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.01601289, + "spread": 0.03562288, + "score": 0.0390564 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.0923988, + "spread": 0.16141034, + "score": 0.18598612 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.5428003, + "spread": 1.17919406, + "score": 1.29812587 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.06812589, + "spread": 0.19686685, + "score": 0.20832113 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.0603787, + "spread": 0.11418187, + "score": 0.12916302 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00033171, + "spread": 0.00745394, + "score": 0.00746132 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00455544, + "spread": 0.00319252, + "score": 0.00556275 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00892199, + "spread": 0.00875758, + "score": 0.01250188 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00144354, + "spread": 0.00262221, + "score": 0.00299329 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00059618, + "spread": 0.00100752, + "score": 0.00117069 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.07953528, + "spread": 0.11480244, + "score": 0.13966195 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00154293, + "spread": 0.00068076, + "score": 0.00168644 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.0076749, + "spread": 0.0, + "score": 0.0076749 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00597658, + "spread": 0.0130448, + "score": 0.01434873 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00373515, + "spread": 0.00331767, + "score": 0.00499583 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00561061, + "spread": 0.00432203, + "score": 0.00708229 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00211368, + "spread": 0.00196792, + "score": 0.00288797 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00905632, + "spread": 0.01291995, + "score": 0.01577789 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00172391, + "spread": 0.01418339, + "score": 0.01428777 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0022034, + "spread": 0.01070219, + "score": 0.01092666 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00030417, + "spread": 0.00283723, + "score": 0.00285349 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00667616, + "spread": 0.01761667, + "score": 0.01883927 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00625187, + "spread": 0.00708088, + "score": 0.00944588 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.00091805, + "spread": 0.04567174, + "score": 0.04568097 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.01233378, + "spread": 0.01557114, + "score": 0.0198641 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00363952, + "spread": 0.00503531, + "score": 0.00621293 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00192093, + "spread": 0.00055578, + "score": 0.00199971 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00490264, + "spread": 0.00928411, + "score": 0.01049907 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01733007, + "spread": 0.01974002, + "score": 0.02626784 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00812841, + "spread": 0.04360184, + "score": 0.04435303 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.10286383, + "spread": 0.10973732, + "score": 0.15041026 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.23402495, + "spread": 0.41557474, + "score": 0.47693819 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.00961822, + "spread": 0.06278808, + "score": 0.06352049 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.16026441, + "spread": 0.12535548, + "score": 0.20346664 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00026636, + "spread": 0.00608917, + "score": 0.00609499 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00794914, + "spread": 0.00882533, + "score": 0.01187751 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.01067665, + "spread": 0.01242377, + "score": 0.01638111 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00196547, + "spread": 0.00487542, + "score": 0.00525669 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00100438, + "spread": 0.0010145, + "score": 0.00142759 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.11166349, + "spread": 0.13366126, + "score": 0.17416678 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00656546, + "spread": 0.00712254, + "score": 0.00968689 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00383745, + "spread": 0.00383745, + "score": 0.00542697 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00311746, + "spread": 0.01825068, + "score": 0.01851502 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00096158, + "spread": 0.0091357, + "score": 0.00918616 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.0020272, + "spread": 0.00643954, + "score": 0.00675109 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00422292, + "spread": 0.00194523, + "score": 0.0046494 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00575992, + "spread": 0.01561659, + "score": 0.01664496 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00122633, + "spread": 0.00412885, + "score": 0.00430712 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0026191, + "spread": 0.00657273, + "score": 0.00707534 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00048978, + "spread": 0.00195744, + "score": 0.00201779 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00329025, + "spread": 0.00908451, + "score": 0.00966199 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00629042, + "spread": 0.00689705, + "score": 0.00933481 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.05255795, + "spread": 0.00380152, + "score": 0.05269525 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00745651, + "spread": 0.01164008, + "score": 0.01382357 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00192273, + "spread": 0.00168341, + "score": 0.00255554 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00152142, + "spread": 0.00187376, + "score": 0.00241365 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.01143144, + "spread": 0.00537287, + "score": 0.01263113 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.02533713, + "spread": 0.00687573, + "score": 0.02625349 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.03293296, + "spread": 0.01392438, + "score": 0.03575568 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.1023284, + "spread": 0.08207299, + "score": 0.13117575 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.21995887, + "spread": 0.10686038, + "score": 0.24454252 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.0042919, + "spread": 0.44634411, + "score": 0.44636474 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.08626944, + "spread": 0.15666804, + "score": 0.17884991 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00113697, + "spread": 0.00300452, + "score": 0.00321245 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.0063365, + "spread": 0.00422148, + "score": 0.00761394 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00625211, + "spread": 0.00440697, + "score": 0.0076492 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00336263, + "spread": 0.00330458, + "score": 0.0047146 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00076722, + "spread": 0.00254555, + "score": 0.00265865 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.13584131, + "spread": 0.0927196, + "score": 0.16446819 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00346358, + "spread": 0.00316556, + "score": 0.00469225 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.01083697, + "spread": 0.0024781, + "score": 0.0111167 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.01040474, + "spread": 0.0085229, + "score": 0.01344985 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00482171, + "spread": 0.00438774, + "score": 0.00651929 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00406475, + "spread": 0.00644837, + "score": 0.00762257 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00095646, + "spread": 0.00347279, + "score": 0.0036021 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00308628, + "spread": 0.00487118, + "score": 0.00576659 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00274139, + "spread": 0.00687668, + "score": 0.00740297 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.0006817, + "spread": 0.00095636, + "score": 0.00117446 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -1.52e-06, + "spread": 0.00039675, + "score": 0.00039676 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00071264, + "spread": 0.00716559, + "score": 0.00720095 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00152621, + "spread": 0.00714364, + "score": 0.00730486 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.01265874, + "spread": 0.0476134, + "score": 0.04926743 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00963878, + "spread": 0.00225502, + "score": 0.00989905 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00189131, + "spread": 0.00274194, + "score": 0.00333096 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00024445, + "spread": 0.00419765, + "score": 0.00420476 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00713136, + "spread": 0.00430492, + "score": 0.00832998 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.02120273, + "spread": 0.00635918, + "score": 0.02213583 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.02013216, + "spread": 0.01308514, + "score": 0.02401093 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07074513, + "spread": 0.04609487, + "score": 0.08443702 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.41078627, + "spread": 0.41306172, + "score": 0.58255072 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.47221773, + "spread": 0.6946849, + "score": 0.83998613 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.10456969, + "spread": 0.08880712, + "score": 0.13719156 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00168685, + "spread": 0.00370504, + "score": 0.00407097 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.003381, + "spread": 0.00209828, + "score": 0.00397919 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.0024736, + "spread": 0.00209432, + "score": 0.00324112 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00078635, + "spread": 0.00068416, + "score": 0.00104232 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00070325, + "spread": 0.00216548, + "score": 0.00227681 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.14118964, + "spread": 0.08812333, + "score": 0.16643388 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00099855, + "spread": 0.01182167, + "score": 0.01186377 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00677894, + "spread": 0.00314573, + "score": 0.00747326 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.00310114, + "spread": 0.00438568, + "score": 0.00537133 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00273189, + "spread": 0.0078202, + "score": 0.00828364 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.0019747, + "spread": 0.00694992, + "score": 0.00722502 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00030219, + "spread": 0.00744431, + "score": 0.00745044 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00444666, + "spread": 0.0115772, + "score": 0.01240179 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.02156958, + "spread": 0.03070039, + "score": 0.03752014 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00026803, + "spread": 0.02548297, + "score": 0.02548438 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.02040232, + "spread": 0.02155565, + "score": 0.02967997 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00336006, + "spread": 0.00236412, + "score": 0.00410841 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.03033452, + "spread": 0.03184295, + "score": 0.04397905 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00861721, + "spread": 0.00981022, + "score": 0.01305744 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.01628118, + "spread": 0.01899887, + "score": 0.02502067 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.01040169, + "spread": 0.0164322, + "score": 0.01944769 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00532525, + "spread": 0.0088218, + "score": 0.01030448 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00067286, + "spread": 0.02232708, + "score": 0.02233722 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01146791, + "spread": 0.04454612, + "score": 0.04599859 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.02887723, + "spread": 0.00860709, + "score": 0.03013265 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.3478321, + "spread": 0.68591266, + "score": 0.76906654 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.03766409, + "spread": 0.0671303, + "score": 0.07697441 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.09473355, + "spread": 0.10582127, + "score": 0.14203023 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.07538137, + "spread": 0.08007619, + "score": 0.10997521 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00481551, + "spread": 0.01878702, + "score": 0.01939436 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.01015667, + "spread": 0.02160839, + "score": 0.02387636 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00477256, + "spread": 0.00949688, + "score": 0.01062864 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00680091, + "spread": 0.01177952, + "score": 0.01360182 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.21860916, + "spread": 0.23266616, + "score": 0.31925462 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.0105696, + "spread": 0.01493437, + "score": 0.01829623 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00535259, + "spread": 0.01371195, + "score": 0.01471964 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.01551555, + "spread": 0.01761969, + "score": 0.02347735 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -9.224e-05, + "spread": 0.00202259, + "score": 0.00202469 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.01770201, + "spread": 0.01134504, + "score": 0.02102549 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00563173, + "spread": 0.00912778, + "score": 0.01072533 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00945897, + "spread": 0.00771742, + "score": 0.01220781 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.0025373, + "spread": 0.00250639, + "score": 0.00356649 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00621793, + "spread": 0.00480209, + "score": 0.00785638 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00238348, + "spread": 0.00449783, + "score": 0.00509033 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.01484423, + "spread": 0.0, + "score": 0.01484423 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00227073, + "spread": 0.01142026, + "score": 0.01164382 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00346786, + "spread": 0.00436936, + "score": 0.00557829 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.0022665, + "spread": 0.00210504, + "score": 0.00309326 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00021155, + "spread": 0.00884097, + "score": 0.0088435 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00463714, + "spread": 0.01767916, + "score": 0.0182772 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00461712, + "spread": 0.03635813, + "score": 0.03665012 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.06488519, + "spread": 0.15889245, + "score": 0.17163012 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.54876814, + "spread": 1.23719547, + "score": 1.35343973 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.5868564, + "spread": 1.33661952, + "score": 1.45977813 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.01378846, + "spread": 0.15894199, + "score": 0.15953896 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00466613, + "spread": 0.00843827, + "score": 0.00964247 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00458755, + "spread": 0.00470696, + "score": 0.00657275 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00941956, + "spread": 0.00917955, + "score": 0.01315265 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00140326, + "spread": 0.00243051, + "score": 0.00280652 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.10566132, + "spread": 0.1322587, + "score": 0.16928284 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.0107594, + "spread": 0.01262949, + "score": 0.01659123 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00026945, + "spread": 0.00464673, + "score": 0.00465454 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00547438, + "spread": 0.00536221, + "score": 0.00766304 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00211136, + "spread": 0.00196081, + "score": 0.00288143 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00819734, + "spread": 0.01240588, + "score": 0.01486951 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00012136, + "spread": 0.01117508, + "score": 0.01117574 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00808478, + "spread": 0.01261675, + "score": 0.01498486 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00025162, + "spread": 0.00344477, + "score": 0.00345395 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00798192, + "spread": 0.01744915, + "score": 0.01918812 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00342653, + "spread": 0.00886646, + "score": 0.00950554 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.02672914, + "spread": 0.01219055, + "score": 0.02937783 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.0066548, + "spread": 0.01176891, + "score": 0.01352012 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00636325, + "spread": 0.00439492, + "score": 0.00773345 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00096049, + "spread": 0.00058323, + "score": 0.0011237 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00015493, + "spread": 0.01002826, + "score": 0.01002945 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00331372, + "spread": 0.02221464, + "score": 0.02246043 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.02750877, + "spread": 0.04404974, + "score": 0.05193372 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.08021484, + "spread": 0.1203293, + "score": 0.14461521 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.25382424, + "spread": 0.40560454, + "score": 0.47847861 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.00903268, + "spread": 0.32503493, + "score": 0.32516041 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.12509488, + "spread": 0.23218152, + "score": 0.26373659 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00360203, + "spread": 0.0073917, + "score": 0.00822264 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00788165, + "spread": 0.00979995, + "score": 0.01257615 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.01181414, + "spread": 0.01273381, + "score": 0.0173702 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00234415, + "spread": 0.00540432, + "score": 0.00589082 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.13921445, + "spread": 0.15372702, + "score": 0.20739494 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00188999, + "spread": 0.01751105, + "score": 0.01761275 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00418007, + "spread": 0.00988696, + "score": 0.01073429 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00270493, + "spread": 0.00740434, + "score": 0.00788295 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00423481, + "spread": 0.00193686, + "score": 0.00465672 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0043769, + "spread": 0.01517585, + "score": 0.01579442 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00026052, + "spread": 0.00349253, + "score": 0.00350223 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00804502, + "spread": 0.00652952, + "score": 0.01036132 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00011228, + "spread": 0.00207289, + "score": 0.00207593 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00128487, + "spread": 0.00943332, + "score": 0.00952042 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00483381, + "spread": 0.00700065, + "score": 0.00850735 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.01872078, + "spread": 0.02121536, + "score": 0.02829416 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00344463, + "spread": 0.01212402, + "score": 0.01260386 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00456983, + "spread": 0.00226617, + "score": 0.00510087 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00225466, + "spread": 0.00205125, + "score": 0.00304814 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00659972, + "spread": 0.00681789, + "score": 0.00948894 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01254008, + "spread": 0.00801545, + "score": 0.01488292 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.01328207, + "spread": 0.01479382, + "score": 0.01988141 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07573607, + "spread": 0.0791303, + "score": 0.10953335 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.13867397, + "spread": 0.21466132, + "score": 0.25555812 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.00170367, + "spread": 0.45045746, + "score": 0.45046068 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.1045341, + "spread": 0.1429685, + "score": 0.17710836 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00326311, + "spread": 0.00287076, + "score": 0.00434616 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00640381, + "spread": 0.00521626, + "score": 0.00825942 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00686944, + "spread": 0.0045361, + "score": 0.00823198 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.0044869, + "spread": 0.002866, + "score": 0.00532412 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00154734, + "spread": 0.00268006, + "score": 0.00309467 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.17798176, + "spread": 0.11864709, + "score": 0.21390334 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00068275, + "spread": 0.00118255, + "score": 0.00136549 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00635902, + "spread": 0.00922311, + "score": 0.01120281 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00142358, + "spread": 0.00503475, + "score": 0.00523214 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00323781, + "spread": 0.0068676, + "score": 0.00759259 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00099214, + "spread": 0.00345376, + "score": 0.00359344 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00409495, + "spread": 0.00445688, + "score": 0.00605247 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00154944, + "spread": 0.00664061, + "score": 0.00681898 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00317057, + "spread": 0.00114633, + "score": 0.00337143 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00058397, + "spread": 0.00018921, + "score": 0.00061386 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00034827, + "spread": 0.00729854, + "score": 0.00730685 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00112313, + "spread": 0.00718264, + "score": 0.00726992 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00020516, + "spread": 0.02985593, + "score": 0.02985663 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.0034837, + "spread": 0.00191158, + "score": 0.0039737 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 3.076e-05, + "spread": 0.00258414, + "score": 0.00258433 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00017772, + "spread": 0.00415501, + "score": 0.00415881 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.0034924, + "spread": 0.00533818, + "score": 0.00637911 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01012957, + "spread": 0.00684269, + "score": 0.01222418 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00270306, + "spread": 0.01399959, + "score": 0.01425816 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.05394713, + "spread": 0.04867401, + "score": 0.07265984 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.40229909, + "spread": 0.44764468, + "score": 0.60185573 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.71321819, + "spread": 0.74368607, + "score": 1.03041213 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.13057929, + "spread": 0.10571637, + "score": 0.16800863 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00221441, + "spread": 0.00292851, + "score": 0.00367148 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00286796, + "spread": 0.00257825, + "score": 0.00385649 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00257105, + "spread": 0.0023172, + "score": 0.00346117 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00121112, + "spread": 0.00088516, + "score": 0.00150011 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00199909, + "spread": 0.00286987, + "score": 0.0034975 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.17875648, + "spread": 0.11173531, + "score": 0.21080479 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00320432, + "spread": 0.0119194, + "score": 0.0123426 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00124193, + "spread": 0.00811732, + "score": 0.00821177 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00085335, + "spread": 0.00748446, + "score": 0.00753296 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00052189, + "spread": 0.00779285, + "score": 0.00781031 + } + ] + }, + "toggle": { + "recorded_utc": "2026-07-16T10:27:18Z", + "git_commit": "4f07d64", + "n_replicates": 4, + "seed": 0, + "campaign_months": [ + 1, + 2, + 3, + 6, + 12 + ], + "profiles": [ + "cp_0pct", + "cp_minus_10pct", + "cp_plus_10pct", + "cp_plus_3pct", + "rated_plus_5pct", + "ti_dependent_cp", + "ws_dependent_cp" + ], + "cells": [ + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.0030467, + "spread": 0.00227078, + "score": 0.00379984 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01278279, + "spread": 0.02543963, + "score": 0.02847059 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00143459, + "spread": 0.00864897, + "score": 0.00876713 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0018796, + "spread": 0.01103415, + "score": 0.0111931 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.0004396, + "spread": 0.0036932, + "score": 0.00371927 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00350659, + "spread": 0.0092041, + "score": 0.00984945 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00935936, + "spread": 0.0089177, + "score": 0.01292761 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00603035, + "spread": 0.00229104, + "score": 0.00645089 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00356293, + "spread": 0.0020875, + "score": 0.00412942 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00200722, + "spread": 0.0040443, + "score": 0.00451501 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00270043, + "spread": 0.00492465, + "score": 0.00561645 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00408902, + "spread": 0.00406295, + "score": 0.00576434 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.02972944, + "spread": 0.04697175, + "score": 0.05558944 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.0030467, + "spread": 0.00227078, + "score": 0.00379984 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.0030467, + "spread": 0.00227078, + "score": 0.00379984 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.0030467, + "spread": 0.00227078, + "score": 0.00379984 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00387822, + "spread": 0.05405929, + "score": 0.05419822 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00379682, + "spread": 0.00882853, + "score": 0.00961035 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00201873, + "spread": 0.00857291, + "score": 0.00880739 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00108133, + "spread": 0.00187293, + "score": 0.00216267 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.00246102, + "spread": 0.0622683, + "score": 0.06231692 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00305769, + "spread": 0.0178378, + "score": 0.01809797 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00503482, + "spread": 0.00478519, + "score": 0.00694604 + }, + { + "profile": "cp_0pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00265869, + "spread": 0.00519853, + "score": 0.00583895 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00120784, + "spread": 0.00160438, + "score": 0.00200821 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01227792, + "spread": 0.02020951, + "score": 0.02364681 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00263183, + "spread": 0.0063296, + "score": 0.00685495 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00203437, + "spread": 0.00417821, + "score": 0.00464716 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00282687, + "spread": 0.00248418, + "score": 0.00376329 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00251174, + "spread": 0.00290099, + "score": 0.00383726 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00497138, + "spread": 0.0069533, + "score": 0.00854769 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 9.926e-05, + "spread": 0.00118466, + "score": 0.00118881 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00401615, + "spread": 0.00277899, + "score": 0.00488388 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.0022845, + "spread": 0.0034432, + "score": 0.00413214 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00264162, + "spread": 0.00562238, + "score": 0.00621203 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00596764, + "spread": 0.00762973, + "score": 0.00968636 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.02463284, + "spread": 0.07404246, + "score": 0.07803245 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.09532865, + "spread": 0.18060757, + "score": 0.20422205 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.00584547, + "spread": 0.0076784, + "score": 0.00965025 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.00120784, + "spread": 0.00160438, + "score": 0.00200821 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.02043367, + "spread": 0.0189057, + "score": 0.02783811 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00329246, + "spread": 0.0018614, + "score": 0.00378221 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00110208, + "spread": 0.00339167, + "score": 0.00356623 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00232987, + "spread": 0.0028243, + "score": 0.00366127 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.0075743, + "spread": 0.05336965, + "score": 0.05390445 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00056585, + "spread": 0.01187027, + "score": 0.01188375 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00040927, + "spread": 0.00366992, + "score": 0.00369267 + }, + { + "profile": "cp_0pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00023926, + "spread": 0.00245137, + "score": 0.00246302 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00178914, + "spread": 0.00181368, + "score": 0.00254764 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00781956, + "spread": 0.01439119, + "score": 0.0163784 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00016427, + "spread": 0.00286328, + "score": 0.00286799 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00179282, + "spread": 0.00057927, + "score": 0.00188408 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -8.798e-05, + "spread": 0.00307999, + "score": 0.00308125 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00488096, + "spread": 0.00574674, + "score": 0.00753981 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00212701, + "spread": 0.00319602, + "score": 0.0038391 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00142626, + "spread": 0.01027563, + "score": 0.01037414 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00031214, + "spread": 0.00251402, + "score": 0.00253332 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00368246, + "spread": 0.00270769, + "score": 0.00457078 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00133642, + "spread": 0.00522982, + "score": 0.00539788 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00250142, + "spread": 0.00648093, + "score": 0.00694691 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00808367, + "spread": 0.03346455, + "score": 0.03442705 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.0059818, + "spread": 0.03312889, + "score": 0.0336646 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.15431917, + "spread": 0.2716436, + "score": 0.31241743 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.00178914, + "spread": 0.00181368, + "score": 0.00254764 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00198679, + "spread": 0.02377948, + "score": 0.02386234 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00058739, + "spread": 0.000638, + "score": 0.00086722 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00128595, + "spread": 0.00265784, + "score": 0.00295259 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00078662, + "spread": 0.00374119, + "score": 0.00382299 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00040369, + "spread": 0.00069922, + "score": 0.00080739 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.00913957, + "spread": 0.03598422, + "score": 0.03712675 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00067989, + "spread": 0.01283871, + "score": 0.0128567 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.0056117, + "spread": 0.00505488, + "score": 0.00755268 + }, + { + "profile": "cp_0pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00149004, + "spread": 0.00402495, + "score": 0.0042919 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00149479, + "spread": 0.00150792, + "score": 0.00212326 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0013837, + "spread": 0.00279903, + "score": 0.00312237 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00019516, + "spread": 0.00363531, + "score": 0.00364055 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00221858, + "spread": 0.0036135, + "score": 0.00424022 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00054374, + "spread": 0.00258253, + "score": 0.00263915 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00230793, + "spread": 0.00254739, + "score": 0.0034374 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00218646, + "spread": 0.00616613, + "score": 0.0065423 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.0022366, + "spread": 0.00184051, + "score": 0.00289653 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00035351, + "spread": 0.01132538, + "score": 0.01133089 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00031592, + "spread": 0.00160758, + "score": 0.00163833 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00295747, + "spread": 0.00132267, + "score": 0.00323977 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00061097, + "spread": 0.00222722, + "score": 0.0023095 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00137316, + "spread": 0.01026518, + "score": 0.01035662 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00345684, + "spread": 0.02063452, + "score": 0.02092207 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.03622114, + "spread": 0.03061605, + "score": 0.04742692 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.05402819, + "spread": 0.17329936, + "score": 0.18152607 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.00149479, + "spread": 0.00150792, + "score": 0.00212326 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00342309, + "spread": 0.01471553, + "score": 0.01510842 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00112113, + "spread": 0.00126619, + "score": 0.00169121 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00060485, + "spread": 0.00222229, + "score": 0.00230313 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00021221, + "spread": 0.00267239, + "score": 0.0026808 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00053106, + "spread": 0.00221394, + "score": 0.00227675 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 7.543e-05, + "spread": 0.00013066, + "score": 0.00015087 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.00835053, + "spread": 0.02793809, + "score": 0.02915936 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00293446, + "spread": 0.00531529, + "score": 0.00607152 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00040016, + "spread": 0.00303578, + "score": 0.00306204 + }, + { + "profile": "cp_0pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00271029, + "spread": 0.00515475, + "score": 0.00582384 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00066441, + "spread": 0.00179039, + "score": 0.0019097 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00127162, + "spread": 0.00547184, + "score": 0.00561765 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00242004, + "spread": 0.00415236, + "score": 0.00480611 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00195454, + "spread": 0.001473, + "score": 0.00244743 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00025556, + "spread": 0.00157692, + "score": 0.00159749 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00314629, + "spread": 0.00316308, + "score": 0.00446142 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00058748, + "spread": 0.00287394, + "score": 0.00293337 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.00129465, + "spread": 0.00163866, + "score": 0.00208838 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00197322, + "spread": 0.00433456, + "score": 0.00476256 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00086924, + "spread": 0.00165398, + "score": 0.00186848 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.0002359, + "spread": 0.00134819, + "score": 0.00136867 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00041122, + "spread": 0.00282112, + "score": 0.00285093 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00526656, + "spread": 0.00740041, + "score": 0.0090831 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00392664, + "spread": 0.03010078, + "score": 0.03035581 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.01353556, + "spread": 0.020119, + "score": 0.02424841 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.03415005, + "spread": 0.23038243, + "score": 0.23289974 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.10076944, + "spread": 0.12946093, + "score": 0.16405674 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00684664, + "spread": 0.03874154, + "score": 0.03934188 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00027457, + "spread": 0.00248186, + "score": 0.002497 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00111796, + "spread": 0.00078667, + "score": 0.001367 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00069043, + "spread": 0.00157445, + "score": 0.00171918 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00034475, + "spread": 0.00117372, + "score": 0.0012233 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00044035, + "spread": 0.00149148, + "score": 0.00155513 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.00184894, + "spread": 0.02169328, + "score": 0.02177193 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00089623, + "spread": 0.0024239, + "score": 0.00258428 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00125256, + "spread": 0.0063314, + "score": 0.00645412 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00273301, + "spread": 0.00328756, + "score": 0.00427521 + }, + { + "profile": "cp_0pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.0003648, + "spread": 0.00303662, + "score": 0.00305845 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00275762, + "spread": 0.00206729, + "score": 0.00344647 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00745963, + "spread": 0.02333601, + "score": 0.0244993 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.0035916, + "spread": 0.0081051, + "score": 0.00886523 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0102023, + "spread": 0.01344143, + "score": 0.0168748 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00128679, + "spread": 0.00399143, + "score": 0.00419373 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00495233, + "spread": 0.00893293, + "score": 0.01021385 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00964386, + "spread": 0.00795669, + "score": 0.01250252 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00269194, + "spread": 0.01340285, + "score": 0.01367051 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00488245, + "spread": 0.00264728, + "score": 0.00555396 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00244785, + "spread": 0.0050895, + "score": 0.00564756 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -4.443e-05, + "spread": 0.00485467, + "score": 0.00485487 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00191469, + "spread": 0.00636208, + "score": 0.00664395 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00611376, + "spread": 0.06051473, + "score": 0.06082278 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.03854336, + "spread": 0.18737337, + "score": 0.19129655 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -1.32165961, + "spread": 2.22277578, + "score": 2.58602326 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 1.99753949, + "spread": 3.58508465, + "score": 4.10402193 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00284071, + "spread": 0.05285708, + "score": 0.05293336 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00158193, + "spread": 0.00920707, + "score": 0.00934199 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00292959, + "spread": 0.01115999, + "score": 0.01153811 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00048312, + "spread": 0.00246021, + "score": 0.0025072 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00048533, + "spread": 0.00062741, + "score": 0.00079321 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00246201, + "spread": 0.00348181, + "score": 0.00426432 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.03637749, + "spread": 0.0664106, + "score": 0.07572113 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00085042, + "spread": 0.00026532, + "score": 0.00089085 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00365186, + "spread": 0.01811648, + "score": 0.01848088 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00695033, + "spread": 0.00673623, + "score": 0.00967905 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00186634, + "spread": 0.00445927, + "score": 0.00483408 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00105375, + "spread": 0.00149002, + "score": 0.00182498 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00906196, + "spread": 0.0189908, + "score": 0.02104209 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00496391, + "spread": 0.00531038, + "score": 0.00726915 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01131577, + "spread": 0.00597765, + "score": 0.01279762 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00644464, + "spread": 0.00306318, + "score": 0.00713558 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00066861, + "spread": 0.00299806, + "score": 0.00307171 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.0066647, + "spread": 0.00515646, + "score": 0.00842658 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.0096, + "spread": 0.01059594, + "score": 0.01429804 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00532454, + "spread": 0.00308198, + "score": 0.00615218 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00182486, + "spread": 0.00401927, + "score": 0.00441414 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00042795, + "spread": 0.00553356, + "score": 0.00555009 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00161622, + "spread": 0.00672922, + "score": 0.00692059 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.01491478, + "spread": 0.06120679, + "score": 0.06299779 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.12098275, + "spread": 0.16224921, + "score": 0.2023898 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.17101319, + "spread": 0.20708441, + "score": 0.2685693 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.27506285, + "spread": 0.38849595, + "score": 0.47601332 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.02611852, + "spread": 0.02501414, + "score": 0.03616469 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00084886, + "spread": 0.00349431, + "score": 0.00359593 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00583889, + "spread": 0.0039016, + "score": 0.00702247 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.0040228, + "spread": 0.00305335, + "score": 0.00505033 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00046326, + "spread": 0.00058458, + "score": 0.00074588 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00156458, + "spread": 0.00270993, + "score": 0.00312916 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.02478495, + "spread": 0.05130189, + "score": 0.05697524 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00033875, + "spread": 0.00024636, + "score": 0.00041886 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.01510048, + "spread": 0.0, + "score": 0.01510048 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00066735, + "spread": 0.01177422, + "score": 0.01179312 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00087402, + "spread": 0.00393025, + "score": 0.00402626 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00143931, + "spread": 0.00293114, + "score": 0.00326545 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00174356, + "spread": 0.00172363, + "score": 0.00245171 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00821912, + "spread": 0.01406215, + "score": 0.01628797 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00793503, + "spread": 0.00146606, + "score": 0.00806932 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00945082, + "spread": 0.00314188, + "score": 0.00995939 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00326202, + "spread": 0.00283529, + "score": 0.004322 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00401121, + "spread": 0.0047678, + "score": 0.00623071 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.0012149, + "spread": 0.00417288, + "score": 0.00434613 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00081487, + "spread": 0.00911758, + "score": 0.00915392 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00169786, + "spread": 0.00226398, + "score": 0.00282991 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00345565, + "spread": 0.0030098, + "score": 0.00458262 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00392213, + "spread": 0.00449947, + "score": 0.00596895 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.0054608, + "spread": 0.00593606, + "score": 0.00806581 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00522875, + "spread": 0.03295288, + "score": 0.03336513 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.01729468, + "spread": 0.04035183, + "score": 0.0439019 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.15955562, + "spread": 0.27602389, + "score": 0.31882155 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.33795177, + "spread": 0.49418743, + "score": 0.59869242 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.01447852, + "spread": 0.02007557, + "score": 0.02475189 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.0023764, + "spread": 0.00159388, + "score": 0.00286143 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.0035605, + "spread": 0.00241834, + "score": 0.00430413 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00062329, + "spread": 0.00314032, + "score": 0.00320157 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00020581, + "spread": 0.00138311, + "score": 0.00139834 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00210724, + "spread": 0.00255566, + "score": 0.00331238 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.02751649, + "spread": 0.04277456, + "score": 0.05086079 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00775178, + "spread": 0.01048521, + "score": 0.01303954 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.00755024, + "spread": 0.00755024, + "score": 0.01067766 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00097601, + "spread": 0.0119598, + "score": 0.01199956 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00462972, + "spread": 0.00446164, + "score": 0.00642966 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00231109, + "spread": 0.00407384, + "score": 0.00468373 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00150011, + "spread": 0.00141091, + "score": 0.00205937 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00084372, + "spread": 0.0030403, + "score": 0.0031552 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00804957, + "spread": 0.00180652, + "score": 0.00824979 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01036081, + "spread": 0.00364067, + "score": 0.01098184 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00225974, + "spread": 0.00271359, + "score": 0.00353129 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00135017, + "spread": 0.00259751, + "score": 0.00292746 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00074289, + "spread": 0.0062238, + "score": 0.00626798 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.04025334, + "spread": 0.01002506, + "score": 0.04148293 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00157502, + "spread": 0.0114258, + "score": 0.01153385 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00064853, + "spread": 0.00156348, + "score": 0.00169265 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00281625, + "spread": 0.00102494, + "score": 0.00299696 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00133173, + "spread": 0.00193999, + "score": 0.00235309 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00505329, + "spread": 0.00891889, + "score": 0.01025096 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00648734, + "spread": 0.01851621, + "score": 0.01961977 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.018046, + "spread": 0.0340999, + "score": 0.03858058 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.01778217, + "spread": 0.28555196, + "score": 0.2861051 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.09364203, + "spread": 0.11463957, + "score": 0.14802385 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.0074845, + "spread": 0.01491804, + "score": 0.01669029 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.0039673, + "spread": 0.00197638, + "score": 0.00443233 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00380749, + "spread": 0.00193924, + "score": 0.00427289 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00137492, + "spread": 0.00247626, + "score": 0.00283236 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 5.42e-05, + "spread": 0.002173, + "score": 0.00217367 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00167374, + "spread": 0.00205042, + "score": 0.00264682 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.02936824, + "spread": 0.04718324, + "score": 0.05557654 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00593647, + "spread": 0.00607219, + "score": 0.00849195 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.01509847, + "spread": 2.01e-06, + "score": 0.01509847 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00297187, + "spread": 0.00477096, + "score": 0.00562086 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 1.41e-05, + "spread": 0.00284851, + "score": 0.00284855 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00348023, + "spread": 0.00497459, + "score": 0.00607112 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00070416, + "spread": 0.00169026, + "score": 0.00183107 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00114147, + "spread": 0.00539885, + "score": 0.0055182 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00427956, + "spread": 0.00350992, + "score": 0.00553481 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01085842, + "spread": 0.00214447, + "score": 0.01106815 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00213027, + "spread": 0.00148185, + "score": 0.00259498 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00201281, + "spread": 0.00252605, + "score": 0.00322991 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00169832, + "spread": 0.00272045, + "score": 0.00320704 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.01368949, + "spread": 0.05131833, + "score": 0.05311283 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00108165, + "spread": 0.00485567, + "score": 0.00497469 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -9.149e-05, + "spread": 0.00150045, + "score": 0.00150324 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00026412, + "spread": 0.00132826, + "score": 0.00135427 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00233635, + "spread": 0.00258497, + "score": 0.00348434 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00887722, + "spread": 0.00749427, + "score": 0.01161762 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00028339, + "spread": 0.02497055, + "score": 0.02497216 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.00516336, + "spread": 0.02020941, + "score": 0.02085859 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.03431905, + "spread": 0.20122926, + "score": 0.20413479 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.58653318, + "spread": 0.54917964, + "score": 0.80350448 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.0019558, + "spread": 0.04354781, + "score": 0.0435917 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00296776, + "spread": 0.00195653, + "score": 0.00355466 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00293869, + "spread": 0.00084934, + "score": 0.00305897 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00215333, + "spread": 0.0014599, + "score": 0.00260156 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00015025, + "spread": 0.00125897, + "score": 0.0012679 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00091609, + "spread": 0.00028754, + "score": 0.00096016 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.02584165, + "spread": 0.03805987, + "score": 0.04600375 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.00370367, + "spread": 0.00388038, + "score": 0.00536419 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.00868406, + "spread": 0.0071939, + "score": 0.01127675 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.00284956, + "spread": 0.00402989, + "score": 0.00493558 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00194596, + "spread": 0.00528781, + "score": 0.00563451 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00213594, + "spread": 0.00333883, + "score": 0.00396359 + }, + { + "profile": "cp_minus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00140791, + "spread": 0.00295373, + "score": 0.00327211 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00333687, + "spread": 0.0024731, + "score": 0.00415342 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01536163, + "spread": 0.0289977, + "score": 0.03281533 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00656457, + "spread": 0.00820794, + "score": 0.01051017 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00141431, + "spread": 0.01155107, + "score": 0.01163733 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00053613, + "spread": 0.00477663, + "score": 0.00480662 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00449718, + "spread": 0.00932383, + "score": 0.01035174 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.01111534, + "spread": 0.00903104, + "score": 0.01432168 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00778187, + "spread": 0.01521853, + "score": 0.01709272 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00250916, + "spread": 0.00312471, + "score": 0.00400745 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00144231, + "spread": 0.00304453, + "score": 0.00336889 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00590612, + "spread": 0.00454287, + "score": 0.00745117 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00750551, + "spread": 0.0030487, + "score": 0.00810106 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.06628005, + "spread": 0.03676379, + "score": 0.07579328 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.03244887, + "spread": 0.19152239, + "score": 0.19425178 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 1.31556511, + "spread": 2.22002001, + "score": 2.58054266 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -2.00363399, + "spread": 3.58505801, + "score": 4.10696848 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.01402062, + "spread": 0.06619859, + "score": 0.06766706 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.0052947, + "spread": 0.00926279, + "score": 0.01066926 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00075103, + "spread": 0.01406488, + "score": 0.01408492 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00140086, + "spread": 0.00083462, + "score": 0.00163064 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00048533, + "spread": 0.00062741, + "score": 0.00079321 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00246201, + "spread": 0.00348181, + "score": 0.00426432 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.03946131, + "spread": 0.08612877, + "score": 0.09473838 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00085042, + "spread": 0.00026532, + "score": 0.00089085 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00252327, + "spread": 0.01674188, + "score": 0.01693096 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.0044257, + "spread": 0.00325574, + "score": 0.00549424 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00287471, + "spread": 0.00613526, + "score": 0.00677535 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00136209, + "spread": 0.00171977, + "score": 0.00219383 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00912097, + "spread": 0.02077693, + "score": 0.02269081 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00888611, + "spread": 0.0080336, + "score": 0.01197922 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00027021, + "spread": 0.00299695, + "score": 0.00300911 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00162122, + "spread": 0.00258974, + "score": 0.00305533 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00036558, + "spread": 0.00379904, + "score": 0.00381659 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.0039065, + "spread": 0.00522269, + "score": 0.00652206 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00828745, + "spread": 0.01420377, + "score": 0.01644473 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00233707, + "spread": 0.0026838, + "score": 0.00355875 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00242433, + "spread": 0.00303469, + "score": 0.00388416 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00595461, + "spread": 0.0052024, + "score": 0.00790711 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00986793, + "spread": 0.00917085, + "score": 0.01347147 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.03416593, + "spread": 0.08574239, + "score": 0.0922988 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07164115, + "spread": 0.2088393, + "score": 0.22078566 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.15863312, + "spread": 0.21690375, + "score": 0.26872235 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.27264701, + "spread": 0.38856537, + "score": 0.47467824 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.01756146, + "spread": 0.02611529, + "score": 0.03147083 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00620892, + "spread": 0.00152961, + "score": 0.00639456 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00358933, + "spread": 0.00371385, + "score": 0.00516488 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00052363, + "spread": 0.00289434, + "score": 0.00294133 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00046326, + "spread": 0.00058458, + "score": 0.00074588 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00156458, + "spread": 0.00270993, + "score": 0.00312916 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.00797098, + "spread": 0.05877844, + "score": 0.05931645 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00033875, + "spread": 0.00024636, + "score": 0.00041886 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.01510048, + "spread": 0.0, + "score": 0.01510048 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00144964, + "spread": 0.01367939, + "score": 0.01375599 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00032951, + "spread": 0.00411166, + "score": 0.00412485 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00091343, + "spread": 0.00267181, + "score": 0.00282363 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00183546, + "spread": 0.00190493, + "score": 0.00264531 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00483991, + "spread": 0.0140561, + "score": 0.01486603 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00621022, + "spread": 0.00426364, + "score": 0.00753296 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00039899, + "spread": 0.00132837, + "score": 0.001387 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00124666, + "spread": 0.00262472, + "score": 0.00290574 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00366217, + "spread": 0.00756357, + "score": 0.00840352 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00251936, + "spread": 0.00412775, + "score": 0.00483586 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.0038951, + "spread": 0.01140621, + "score": 0.01205294 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00088597, + "spread": 0.00297391, + "score": 0.00310308 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00380193, + "spread": 0.00273862, + "score": 0.00468559 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00066211, + "spread": 0.00559378, + "score": 0.00563283 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00033382, + "spread": 0.00625215, + "score": 0.00626106 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00968845, + "spread": 0.0343033, + "score": 0.03564523 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.00529893, + "spread": 0.03435914, + "score": 0.03476534 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.50978738, + "spread": 0.878815, + "score": 1.01597194 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.13090984, + "spread": 0.14646347, + "score": 0.19644066 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00317597, + "spread": 0.03158858, + "score": 0.03174783 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00337781, + "spread": 0.00121442, + "score": 0.00358948 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00592268, + "spread": 0.00313242, + "score": 0.00670001 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00253987, + "spread": 0.0035502, + "score": 0.00436519 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00061793, + "spread": 0.00052631, + "score": 0.00081169 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00210724, + "spread": 0.00255566, + "score": 0.00331238 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.00843796, + "spread": 0.03952632, + "score": 0.04041694 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00775178, + "spread": 0.01048521, + "score": 0.01303954 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00755024, + "spread": 0.00755024, + "score": 0.01067766 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00048842, + "spread": 0.01334919, + "score": 0.01335812 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00691795, + "spread": 0.0055751, + "score": 0.00888481 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00012117, + "spread": 0.00418388, + "score": 0.00418563 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00148947, + "spread": 0.00160507, + "score": 0.00218969 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.0006485, + "spread": 0.00240466, + "score": 0.00249057 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00389861, + "spread": 0.00537753, + "score": 0.00664207 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00016891, + "spread": 0.00272308, + "score": 0.00272831 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00139385, + "spread": 0.00263102, + "score": 0.00297743 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00042662, + "spread": 0.00279452, + "score": 0.00282689 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00209317, + "spread": 0.00671154, + "score": 0.00703037 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.03578014, + "spread": 0.00634404, + "score": 0.03633821 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00090917, + "spread": 0.01078805, + "score": 0.01082629 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00149652, + "spread": 0.00178, + "score": 0.00232551 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00287734, + "spread": 0.00211244, + "score": 0.00356952 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00234408, + "spread": 0.00262644, + "score": 0.00352036 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.0006818, + "spread": 0.00947384, + "score": 0.00949835 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00033214, + "spread": 0.02365879, + "score": 0.02366112 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.05299862, + "spread": 0.02907164, + "score": 0.06044844 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.11595501, + "spread": 0.35105664, + "score": 0.36971114 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.0966316, + "spread": 0.11455271, + "score": 0.14986657 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00245347, + "spread": 0.01781811, + "score": 0.01798623 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00211762, + "spread": 0.00120091, + "score": 0.00243444 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00521216, + "spread": 0.00238214, + "score": 0.00573072 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00163669, + "spread": 0.00286831, + "score": 0.00330242 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00092141, + "spread": 0.00233465, + "score": 0.0025099 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.0016017, + "spread": 0.00205821, + "score": 0.002608 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.04107019, + "spread": 0.04617055, + "score": 0.06179385 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00593647, + "spread": 0.00607219, + "score": 0.00849195 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.01509847, + "spread": 2.01e-06, + "score": 0.01509847 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00241317, + "spread": 0.00592787, + "score": 0.00640024 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00102805, + "spread": 0.00297736, + "score": 0.00314985 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00171104, + "spread": 0.00550972, + "score": 0.00576929 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00062469, + "spread": 0.00189072, + "score": 0.00199124 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00111969, + "spread": 0.00624481, + "score": 0.00634439 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.0068226, + "spread": 0.0034169, + "score": 0.00763041 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.0005189, + "spread": 0.00184486, + "score": 0.00191644 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00068179, + "spread": 0.001395, + "score": 0.0015527 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00103044, + "spread": 0.00360106, + "score": 0.00374559 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00013382, + "spread": 0.00353334, + "score": 0.00353587 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.01110019, + "spread": 0.04924013, + "score": 0.05047578 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00214417, + "spread": 0.00412826, + "score": 0.00465188 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00175795, + "spread": 0.00179021, + "score": 0.00250903 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00026284, + "spread": 0.00167903, + "score": 0.00169948 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00150078, + "spread": 0.00350855, + "score": 0.00381605 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00166435, + "spread": 0.00738132, + "score": 0.00756664 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.01027979, + "spread": 0.03426852, + "score": 0.03577717 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.02132593, + "spread": 0.02334229, + "score": 0.03161737 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.02132215, + "spread": 0.26300201, + "score": 0.26386491 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.5798823, + "spread": 0.67024434, + "score": 0.88627927 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.0147072, + "spread": 0.04312157, + "score": 0.04556063 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00377844, + "spread": 0.00289765, + "score": 0.00476161 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00512689, + "spread": 0.00101546, + "score": 0.00522648 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00049162, + "spread": 0.00150669, + "score": 0.00158487 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00073238, + "spread": 0.00113368, + "score": 0.00134967 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00213095, + "spread": 0.0030892, + "score": 0.00375288 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.02537676, + "spread": 0.02167905, + "score": 0.03337605 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00541189, + "spread": 0.00432498, + "score": 0.00692777 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00868406, + "spread": 0.0071939, + "score": 0.01127675 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.00284956, + "spread": 0.00402989, + "score": 0.00493558 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00108328, + "spread": 0.00685938, + "score": 0.0069444 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00320891, + "spread": 0.00337448, + "score": 0.00465663 + }, + { + "profile": "cp_plus_10pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00078013, + "spread": 0.00283918, + "score": 0.00294441 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00313355, + "spread": 0.00233166, + "score": 0.00390586 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0146248, + "spread": 0.02683313, + "score": 0.03055981 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00186223, + "spread": 0.00979943, + "score": 0.0099748 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00089199, + "spread": 0.01003881, + "score": 0.01007836 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00047829, + "spread": 0.00426014, + "score": 0.00428691 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00431282, + "spread": 0.00880221, + "score": 0.009802 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00984242, + "spread": 0.01037075, + "score": 0.01429775 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00595354, + "spread": 0.00583936, + "score": 0.00833923 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.0032031, + "spread": 0.0023631, + "score": 0.00398046 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00188656, + "spread": 0.00381876, + "score": 0.00425934 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00365827, + "spread": 0.00484998, + "score": 0.00607497 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00497894, + "spread": 0.00324992, + "score": 0.00594575 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.04107508, + "spread": 0.04426963, + "score": 0.06039009 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.00760218, + "spread": 0.05891622, + "score": 0.05940467 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.39253705, + "spread": 0.66504363, + "score": 0.7722489 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.60322268, + "spread": 1.0755106, + "score": 1.23312637 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.01201291, + "spread": 0.06820977, + "score": 0.06925953 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00432734, + "spread": 0.00871078, + "score": 0.00972644 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00187661, + "spread": 0.00974741, + "score": 0.00992641 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00123769, + "spread": 0.00166015, + "score": 0.00207074 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.0001456, + "spread": 0.00018822, + "score": 0.00023796 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.0007386, + "spread": 0.00104454, + "score": 0.0012793 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.01160818, + "spread": 0.06112455, + "score": 0.06221704 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00025513, + "spread": 7.959e-05, + "score": 0.00026725 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00241164, + "spread": 0.01846956, + "score": 0.01862634 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00516041, + "spread": 0.00425103, + "score": 0.00668589 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00259278, + "spread": 0.00577687, + "score": 0.00633204 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00125412, + "spread": 0.00163889, + "score": 0.00206367 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01092687, + "spread": 0.02017354, + "score": 0.02294272 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00496243, + "spread": 0.00541533, + "score": 0.00734517 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00129276, + "spread": 0.00398678, + "score": 0.00419114 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00292022, + "spread": 0.00329616, + "score": 0.00440368 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00228452, + "spread": 0.00289076, + "score": 0.0036845 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00385383, + "spread": 0.00680866, + "score": 0.00782367 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00270556, + "spread": 0.00461371, + "score": 0.00534849 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00354281, + "spread": 0.00284035, + "score": 0.00454083 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00236789, + "spread": 0.00354185, + "score": 0.00426047 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00363501, + "spread": 0.00499944, + "score": 0.00618123 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00663927, + "spread": 0.00750441, + "score": 0.01001978 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.0298837, + "spread": 0.07744879, + "score": 0.08301415 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.07933683, + "spread": 0.17786849, + "score": 0.19476019 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.04352682, + "spread": 0.06848816, + "score": 0.08114932 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.08094861, + "spread": 0.11660392, + "score": 0.14194771 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.02130705, + "spread": 0.01966394, + "score": 0.02899415 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00409472, + "spread": 0.00175895, + "score": 0.00445652 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00011018, + "spread": 0.00351586, + "score": 0.00351759 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00170508, + "spread": 0.00312859, + "score": 0.00356306 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00013898, + "spread": 0.00017537, + "score": 0.00022376 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00046937, + "spread": 0.00081298, + "score": 0.00093875 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.00168437, + "spread": 0.05339005, + "score": 0.05341661 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00010162, + "spread": 7.391e-05, + "score": 0.00012566 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00453015, + "spread": 0.0, + "score": 0.00453015 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00124987, + "spread": 0.01181842, + "score": 0.01188433 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00056103, + "spread": 0.00376058, + "score": 0.0038022 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -4.991e-05, + "spread": 0.00230403, + "score": 0.00230457 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00180323, + "spread": 0.00184107, + "score": 0.00257705 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00836838, + "spread": 0.01515533, + "score": 0.01731224 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00336162, + "spread": 0.00332948, + "score": 0.00473138 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00057749, + "spread": 0.00050018, + "score": 0.00076398 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00069242, + "spread": 0.00313425, + "score": 0.00320982 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00492627, + "spread": 0.00676362, + "score": 0.00836748 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00314783, + "spread": 0.00371753, + "score": 0.00487123 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00177451, + "spread": 0.01019567, + "score": 0.01034894 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 5.367e-05, + "spread": 0.00258889, + "score": 0.00258944 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00372479, + "spread": 0.0027198, + "score": 0.00461209 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00084001, + "spread": 0.00497078, + "score": 0.00504125 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00112254, + "spread": 0.00622563, + "score": 0.00632602 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00767417, + "spread": 0.03297709, + "score": 0.03385825 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.0029632, + "spread": 0.04035468, + "score": 0.04046333 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.25588611, + "spread": 0.45185537, + "score": 0.51927929 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.10371183, + "spread": 0.14868284, + "score": 0.1812808 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00178213, + "spread": 0.02489506, + "score": 0.02495876 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00157696, + "spread": 0.00082138, + "score": 0.00177805 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00257405, + "spread": 0.00295079, + "score": 0.00391572 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00128241, + "spread": 0.00378197, + "score": 0.00399348 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00042041, + "spread": 0.00047419, + "score": 0.00063372 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00063217, + "spread": 0.0007667, + "score": 0.00099371 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.00571813, + "spread": 0.03661746, + "score": 0.03706123 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00232553, + "spread": 0.00314556, + "score": 0.00391186 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00226507, + "spread": 0.00226507, + "score": 0.0032033 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00093851, + "spread": 0.01299199, + "score": 0.01302585 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00610035, + "spread": 0.00522719, + "score": 0.00803354 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00112161, + "spread": 0.0043043, + "score": 0.00444803 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00149298, + "spread": 0.00153721, + "score": 0.00214289 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00030001, + "spread": 0.00260377, + "score": 0.002621 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00099712, + "spread": 0.00448475, + "score": 0.00459426 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00088231, + "spread": 0.00319765, + "score": 0.00331714 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00111293, + "spread": 0.0025814, + "score": 0.0028111 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00208812, + "spread": 0.00261735, + "score": 0.00334825 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00207184, + "spread": 0.00659866, + "score": 0.00691627 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.00916885, + "spread": 0.00061443, + "score": 0.00918941 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -2.6e-07, + "spread": 0.01048645, + "score": 0.01048645 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00064749, + "spread": 0.00164407, + "score": 0.00176698 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00286744, + "spread": 0.00155175, + "score": 0.00326039 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00085053, + "spread": 0.00229437, + "score": 0.00244694 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00087381, + "spread": 0.00979282, + "score": 0.00983173 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00229511, + "spread": 0.02187709, + "score": 0.02199715 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.04305858, + "spread": 0.02984954, + "score": 0.05239309 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.07214575, + "spread": 0.20350397, + "score": 0.21591405 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.03003562, + "spread": 0.03436552, + "score": 0.04564129 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00393136, + "spread": 0.01556319, + "score": 0.01605205 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00018979, + "spread": 0.00109564, + "score": 0.00111195 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00218473, + "spread": 0.00214066, + "score": 0.00305867 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00059953, + "spread": 0.00266338, + "score": 0.00273003 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00070364, + "spread": 0.00229128, + "score": 0.00239688 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.0004253, + "spread": 0.00063351, + "score": 0.00076303 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.01898044, + "spread": 0.03251592, + "score": 0.03765026 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00178094, + "spread": 0.00182166, + "score": 0.00254758 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00452954, + "spread": 6e-07, + "score": 0.00452954 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00241685, + "spread": 0.0054525, + "score": 0.00596414 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00072031, + "spread": 0.00281083, + "score": 0.00290165 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00233771, + "spread": 0.00553958, + "score": 0.00601265 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00065249, + "spread": 0.00182048, + "score": 0.00193387 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0007528, + "spread": 0.00609566, + "score": 0.00614197 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00340901, + "spread": 0.00381316, + "score": 0.00511483 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -4.433e-05, + "spread": 0.00155955, + "score": 0.00156018 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00058363, + "spread": 0.00150839, + "score": 0.00161737 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00289113, + "spread": 0.0030776, + "score": 0.00422259 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00079355, + "spread": 0.00318295, + "score": 0.00328038 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.0024238, + "spread": 0.01409659, + "score": 0.01430345 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00157778, + "spread": 0.0043806, + "score": 0.00465607 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00116618, + "spread": 0.00159557, + "score": 0.00197632 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00023041, + "spread": 0.00143503, + "score": 0.00145341 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.0001521, + "spread": 0.00314094, + "score": 0.00314463 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.0041865, + "spread": 0.00741872, + "score": 0.00851846 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00597612, + "spread": 0.0309225, + "score": 0.03149468 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.014087, + "spread": 0.01971759, + "score": 0.02423277 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.02608971, + "spread": 0.24291303, + "score": 0.24431008 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.1088106, + "spread": 0.14973697, + "score": 0.18509702 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.01040754, + "spread": 0.0481373, + "score": 0.04924953 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00137916, + "spread": 0.00255668, + "score": 0.00290495 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00237116, + "spread": 0.00088799, + "score": 0.00253198 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00036279, + "spread": 0.00157332, + "score": 0.00161461 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00048286, + "spread": 0.00115074, + "score": 0.00124794 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00102724, + "spread": 0.00198108, + "score": 0.00223157 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.0042678, + "spread": 0.01906216, + "score": 0.01953407 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00225854, + "spread": 0.00261186, + "score": 0.00345294 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00260522, + "spread": 0.00215817, + "score": 0.00338303 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.00085487, + "spread": 0.00120897, + "score": 0.00148067 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00119143, + "spread": 0.00627168, + "score": 0.00638384 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00306532, + "spread": 0.00329207, + "score": 0.00449821 + }, + { + "profile": "cp_plus_3pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -5.835e-05, + "spread": 0.00303271, + "score": 0.00303327 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.0030756, + "spread": 0.00229937, + "score": 0.0038401 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01478914, + "spread": 0.02672256, + "score": 0.030542 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00185144, + "spread": 0.01054317, + "score": 0.0107045 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00593935, + "spread": 0.01207745, + "score": 0.01345886 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.01320231, + "spread": 0.02049543, + "score": 0.02437958 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00313941, + "spread": 0.0075615, + "score": 0.00818732 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00806057, + "spread": 0.00967871, + "score": 0.01259564 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00420255, + "spread": 0.00629493, + "score": 0.00756886 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00443799, + "spread": 0.00244783, + "score": 0.0050683 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00229214, + "spread": 0.00455254, + "score": 0.00509701 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00145451, + "spread": 0.00539022, + "score": 0.00558301 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00254552, + "spread": 0.00380516, + "score": 0.00457809 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.01862515, + "spread": 0.05467725, + "score": 0.05776243 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.01068045, + "spread": 0.00561563, + "score": 0.01206679 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.01068016, + "spread": 0.00559998, + "score": 0.01205926 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.01068046, + "spread": 0.00562224, + "score": 0.01206987 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00194484, + "spread": 0.06170374, + "score": 0.06173438 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00311572, + "spread": 0.01091478, + "score": 0.01135078 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00643615, + "spread": 0.02209322, + "score": 0.02301161 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.03691403, + "spread": 0.02152682, + "score": 0.0427323 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.04937198, + "spread": 0.00035907, + "score": 0.04937329 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.04834897, + "spread": 0.00170345, + "score": 0.04837897 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.00313805, + "spread": 0.06137769, + "score": 0.06145785 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.04945674, + "spread": 1.461e-05, + "score": 0.04945674 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00283884, + "spread": 0.01843828, + "score": 0.01865554 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00101614, + "spread": 0.00393623, + "score": 0.00406527 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00331063, + "spread": 0.00643391, + "score": 0.00723571 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00121559, + "spread": 0.00162645, + "score": 0.00203051 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01071882, + "spread": 0.02044472, + "score": 0.02308419 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.0005071, + "spread": 0.00685228, + "score": 0.00687102 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0070095, + "spread": 0.00458668, + "score": 0.0083768 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00540974, + "spread": 0.00344571, + "score": 0.00641391 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00299792, + "spread": 0.00354939, + "score": 0.00464604 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00544646, + "spread": 0.00656992, + "score": 0.00853392 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00526792, + "spread": 0.00426944, + "score": 0.00678079 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00493712, + "spread": 0.00336329, + "score": 0.00597385 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.0023152, + "spread": 0.00414419, + "score": 0.00474705 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00140177, + "spread": 0.00549256, + "score": 0.00566861 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.0042255, + "spread": 0.00745338, + "score": 0.00856783 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.02162857, + "spread": 0.07565648, + "score": 0.07868734 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.10467555, + "spread": 0.18869229, + "score": 0.21578172 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.00601488, + "spread": 0.01196043, + "score": 0.01338771 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.01402895, + "spread": 0.00364216, + "score": 0.01449403 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.03103419, + "spread": 0.02656573, + "score": 0.04085167 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00029731, + "spread": 0.00196638, + "score": 0.00198873 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00097953, + "spread": 0.00380047, + "score": 0.00392467 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00092485, + "spread": 0.00234195, + "score": 0.00251795 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.04931545, + "spread": 0.00031108, + "score": 0.04931643 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.04880387, + "spread": 0.00129025, + "score": 0.04882092 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.01088979, + "spread": 0.05576764, + "score": 0.05682092 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.04956249, + "spread": 9.114e-05, + "score": 0.04956257 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.0421986, + "spread": 0.0, + "score": 0.0421986 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00249177, + "spread": 0.01326539, + "score": 0.01349739 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00160145, + "spread": 0.00462055, + "score": 0.0048902 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00299715, + "spread": 0.00237329, + "score": 0.00382301 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00182837, + "spread": 0.00184695, + "score": 0.00259887 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00918062, + "spread": 0.01412632, + "score": 0.01684746 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00186504, + "spread": 0.00331884, + "score": 0.00380697 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00520441, + "spread": 0.00135889, + "score": 0.0053789 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00179325, + "spread": 0.00312091, + "score": 0.00359941 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00562239, + "spread": 0.0055063, + "score": 0.0078696 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00194181, + "spread": 0.00390706, + "score": 0.004363 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -5.812e-05, + "spread": 0.01093513, + "score": 0.01093529 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00094836, + "spread": 0.00241172, + "score": 0.00259148 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00367147, + "spread": 0.00289191, + "score": 0.00467363 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00276045, + "spread": 0.0051827, + "score": 0.00587201 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00517839, + "spread": 0.00679705, + "score": 0.00854491 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00513483, + "spread": 0.03580837, + "score": 0.03617466 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.01006111, + "spread": 0.03402578, + "score": 0.0354821 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.14487552, + "spread": 0.27736542, + "score": 0.3129225 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.01786503, + "spread": 0.00127409, + "score": 0.0179104 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00805019, + "spread": 0.02233219, + "score": 0.02373884 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00256097, + "spread": 0.00096549, + "score": 0.00273692 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00100167, + "spread": 0.00248266, + "score": 0.00267711 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00185566, + "spread": 0.00328251, + "score": 0.00377073 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.03724103, + "spread": 0.02084082, + "score": 0.04267592 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.04858036, + "spread": 0.00120721, + "score": 0.04859536 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.01450379, + "spread": 0.03797868, + "score": 0.04065391 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.04602255, + "spread": 0.0050346, + "score": 0.04629711 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.04586812, + "spread": 0.00366953, + "score": 0.04601467 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00230543, + "spread": 0.01383628, + "score": 0.01402703 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00741029, + "spread": 0.00574957, + "score": 0.00937923 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00371152, + "spread": 0.00458761, + "score": 0.00590098 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00153543, + "spread": 0.0015354, + "score": 0.0021714 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00151446, + "spread": 0.002269, + "score": 0.00272799 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00334162, + "spread": 0.00313303, + "score": 0.00458064 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00449783, + "spread": 0.00351429, + "score": 0.00570795 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00060612, + "spread": 0.00260506, + "score": 0.00267465 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00310049, + "spread": 0.00262227, + "score": 0.00406071 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00180246, + "spread": 0.0057323, + "score": 0.00600901 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.01746784, + "spread": 0.0020415, + "score": 0.01758674 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00134274, + "spread": 0.01163363, + "score": 0.01171086 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -6.416e-05, + "spread": 0.00150607, + "score": 0.00150743 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.0030296, + "spread": 0.00119119, + "score": 0.00325537 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00015251, + "spread": 0.00228695, + "score": 0.00229203 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00333565, + "spread": 0.01044247, + "score": 0.01096228 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00287107, + "spread": 0.0193133, + "score": 0.01952554 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.03605172, + "spread": 0.03018668, + "score": 0.04702087 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.06128761, + "spread": 0.16897928, + "score": 0.17975029 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.01848281, + "spread": 0.0031539, + "score": 0.01874997 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00508538, + "spread": 0.01581023, + "score": 0.01660796 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00317197, + "spread": 0.00161025, + "score": 0.00355729 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.0009096, + "spread": 0.00179299, + "score": 0.00201052 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00018171, + "spread": 0.00242747, + "score": 0.00243426 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00108271, + "spread": 0.00210152, + "score": 0.00236404 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.0361096, + "spread": 0.02192102, + "score": 0.04224257 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.00430704, + "spread": 0.0267748, + "score": 0.027119 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.04679427, + "spread": 0.00299786, + "score": 0.0468902 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.04224692, + "spread": 4.832e-05, + "score": 0.04224695 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -2e-08, + "spread": 0.0, + "score": 2e-08 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00352032, + "spread": 0.00507414, + "score": 0.00617572 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00092768, + "spread": 0.00324281, + "score": 0.00337289 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00379758, + "spread": 0.00536257, + "score": 0.00657106 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00069611, + "spread": 0.00182653, + "score": 0.00195468 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00216346, + "spread": 0.00561145, + "score": 0.00601406 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00023649, + "spread": 0.00418962, + "score": 0.00419628 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0037074, + "spread": 0.00196085, + "score": 0.00419401 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00042569, + "spread": 0.00159993, + "score": 0.00165559 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00374255, + "spread": 0.00325439, + "score": 0.00495961 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00127113, + "spread": 0.00280386, + "score": 0.00307854 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": 0.02158696, + "spread": 0.00062017, + "score": 0.02159587 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00136964, + "spread": 0.00482067, + "score": 0.00501146 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00038169, + "spread": 0.00160542, + "score": 0.00165017 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00023744, + "spread": 0.00137839, + "score": 0.00139869 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00160729, + "spread": 0.00290891, + "score": 0.00332342 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00754028, + "spread": 0.00765317, + "score": 0.01074369 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00308575, + "spread": 0.02933894, + "score": 0.02950076 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.01021607, + "spread": 0.0208989, + "score": 0.02326224 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.03343226, + "spread": 0.23164288, + "score": 0.23404303 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.07642297, + "spread": 0.11039854, + "score": 0.13426954 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.01236254, + "spread": 0.04065073, + "score": 0.04248899 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00149205, + "spread": 0.00247647, + "score": 0.00289121 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": -0.00060498, + "spread": 0.00096989, + "score": 0.0011431 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00116514, + "spread": 0.00167345, + "score": 0.00203911 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00034439, + "spread": 0.00144639, + "score": 0.00148682 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00031721, + "spread": 0.00083464, + "score": 0.00089289 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.00370617, + "spread": 0.02080587, + "score": 0.02113339 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.02356071, + "spread": 0.02256018, + "score": 0.03262006 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.04559027, + "spread": 0.00362919, + "score": 0.0457345 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.0152419, + "spread": 0.02155527, + "score": 0.02639972 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00176493, + "spread": 0.00585286, + "score": 0.00611318 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00299885, + "spread": 0.00339284, + "score": 0.00452819 + }, + { + "profile": "rated_plus_5pct", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00102738, + "spread": 0.00332194, + "score": 0.00347718 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00319538, + "spread": 0.00238353, + "score": 0.00398643 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01355147, + "spread": 0.02569807, + "score": 0.02905225 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00413457, + "spread": 0.00740385, + "score": 0.00848008 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00077662, + "spread": 0.0103144, + "score": 0.01034359 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.0005711, + "spread": 0.00454388, + "score": 0.00457963 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00384801, + "spread": 0.00936032, + "score": 0.01012041 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00922264, + "spread": 0.00897444, + "score": 0.01286848 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.02475551, + "spread": 0.02287859, + "score": 0.03370853 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00597225, + "spread": 0.00277245, + "score": 0.00658439 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00284303, + "spread": 0.00375503, + "score": 0.00470989 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00068278, + "spread": 0.00586672, + "score": 0.00590632 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00764922, + "spread": 0.00479046, + "score": 0.00902546 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00672577, + "spread": 0.05639263, + "score": 0.0567923 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.04070709, + "spread": 0.00786051, + "score": 0.04145907 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.04070709, + "spread": 0.00786051, + "score": 0.04145907 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.04070709, + "spread": 0.00786051, + "score": 0.04145907 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.01689397, + "spread": 0.06031245, + "score": 0.06263384 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00521103, + "spread": 0.00900099, + "score": 0.01040061 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00093986, + "spread": 0.01163952, + "score": 0.0116774 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00144112, + "spread": 0.00145728, + "score": 0.00204951 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00020347, + "spread": 0.00025669, + "score": 0.00032755 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00087854, + "spread": 0.00124244, + "score": 0.00152167 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.01204854, + "spread": 0.05961413, + "score": 0.0608195 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00062043, + "spread": 0.00018256, + "score": 0.00064674 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00188407, + "spread": 0.01697078, + "score": 0.01707504 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00472689, + "spread": 0.004717, + "score": 0.00667784 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.0024182, + "spread": 0.00601484, + "score": 0.00648275 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00129, + "spread": 0.00167909, + "score": 0.00211741 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01214556, + "spread": 0.0198645, + "score": 0.02328332 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00612724, + "spread": 0.00621449, + "score": 0.00872714 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00097568, + "spread": 0.00277692, + "score": 0.00294334 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00257132, + "spread": 0.00310943, + "score": 0.00403488 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00220205, + "spread": 0.00315208, + "score": 0.00384508 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00368803, + "spread": 0.00463304, + "score": 0.00592171 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.02435963, + "spread": 0.02508437, + "score": 0.03496594 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00526507, + "spread": 0.0025222, + "score": 0.00583802 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00160853, + "spread": 0.00367129, + "score": 0.00400821 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00053596, + "spread": 0.00569485, + "score": 0.00572001 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00478334, + "spread": 0.00462228, + "score": 0.00665175 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00583272, + "spread": 0.07780393, + "score": 0.07802225 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.11983152, + "spread": 0.17143549, + "score": 0.20916434 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.03152478, + "spread": 0.02074932, + "score": 0.03774051 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.04137209, + "spread": 0.00609443, + "score": 0.04181856 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.02203969, + "spread": 0.02171799, + "score": 0.03094219 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00543748, + "spread": 0.00174667, + "score": 0.00571113 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00144017, + "spread": 0.00418861, + "score": 0.00442928 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.0013594, + "spread": 0.003147, + "score": 0.00342806 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.0002114, + "spread": 0.00022852, + "score": 0.00031131 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.0005705, + "spread": 0.00098813, + "score": 0.00114099 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.01116098, + "spread": 0.05631046, + "score": 0.05740588 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00025218, + "spread": 0.00018569, + "score": 0.00031317 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.01203016, + "spread": 0.0, + "score": 0.01203016 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00168534, + "spread": 0.01298232, + "score": 0.01309126 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00036062, + "spread": 0.00378958, + "score": 0.0038067 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00022775, + "spread": 0.00277272, + "score": 0.00278206 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00182779, + "spread": 0.0018808, + "score": 0.00262264 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00712037, + "spread": 0.01462503, + "score": 0.01626625 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00403999, + "spread": 0.00452272, + "score": 0.00606436 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00164356, + "spread": 0.00106104, + "score": 0.0019563 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00090395, + "spread": 0.00304782, + "score": 0.00317905 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00596894, + "spread": 0.00583542, + "score": 0.00834747 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00254026, + "spread": 0.00432635, + "score": 0.00501699 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00506049, + "spread": 0.01359558, + "score": 0.01450684 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00181054, + "spread": 0.00268386, + "score": 0.00323746 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00312489, + "spread": 0.00291199, + "score": 0.00427137 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00435328, + "spread": 0.00566166, + "score": 0.00714181 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01338806, + "spread": 0.00540999, + "score": 0.01443981 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.00826577, + "spread": 0.0295407, + "score": 0.03067533 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.02484169, + "spread": 0.04097759, + "score": 0.04791944 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.12702858, + "spread": 0.28316369, + "score": 0.31035131 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.04386513, + "spread": 0.00531974, + "score": 0.04418653 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00109253, + "spread": 0.02347432, + "score": 0.02349973 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00286849, + "spread": 0.00099094, + "score": 0.00303483 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00418182, + "spread": 0.00315645, + "score": 0.00523935 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00188453, + "spread": 0.00375254, + "score": 0.00419916 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00041425, + "spread": 0.00036013, + "score": 0.00054891 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00091043, + "spread": 0.00096716, + "score": 0.00132826 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": 0.00921032, + "spread": 0.03671825, + "score": 0.03785577 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00608012, + "spread": 0.0082431, + "score": 0.01024288 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00601508, + "spread": 0.00601508, + "score": 0.00850661 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00170591, + "spread": 0.01277076, + "score": 0.0128842 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00633656, + "spread": 0.00533368, + "score": 0.00828252 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00068106, + "spread": 0.00420468, + "score": 0.00425948 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00149274, + "spread": 0.00157831, + "score": 0.0021724 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00138735, + "spread": 0.00268416, + "score": 0.0030215 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00200405, + "spread": 0.00527101, + "score": 0.00563912 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00136334, + "spread": 0.00278017, + "score": 0.00309645 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00121929, + "spread": 0.002557, + "score": 0.00283283 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00139595, + "spread": 0.00284444, + "score": 0.00316852 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00295415, + "spread": 0.00641998, + "score": 0.00706705 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.06091945, + "spread": 0.00952155, + "score": 0.06165906 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00411951, + "spread": 0.01151784, + "score": 0.01223237 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00132276, + "spread": 0.00191204, + "score": 0.00232499 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00232, + "spread": 0.00143364, + "score": 0.00272722 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00278105, + "spread": 0.00231529, + "score": 0.00361868 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01281398, + "spread": 0.0083279, + "score": 0.01528241 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.02043173, + "spread": 0.02205106, + "score": 0.03006169 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.01756649, + "spread": 0.02741144, + "score": 0.03255716 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.07384845, + "spread": 0.1684744, + "score": 0.18394895 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.04288149, + "spread": 0.00474671, + "score": 0.0431434 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00130036, + "spread": 0.01875255, + "score": 0.01879758 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00137204, + "spread": 0.00093365, + "score": 0.00165958 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00323641, + "spread": 0.00242796, + "score": 0.0040459 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.0009763, + "spread": 0.00269141, + "score": 0.00286301 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00072431, + "spread": 0.00219193, + "score": 0.0023085 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00064002, + "spread": 0.00075377, + "score": 0.00098883 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.01230979, + "spread": 0.03187957, + "score": 0.03417364 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00437423, + "spread": 0.00437593, + "score": 0.0061873 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.01242973, + "spread": 0.00039957, + "score": 0.01243615 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00316498, + "spread": 0.00543784, + "score": 0.00629184 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00092912, + "spread": 0.0030107, + "score": 0.00315081 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00239937, + "spread": 0.00547424, + "score": 0.00597698 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00064343, + "spread": 0.00185842, + "score": 0.00196665 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00169042, + "spread": 0.00624879, + "score": 0.00647339 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00466422, + "spread": 0.00393206, + "score": 0.0061005 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00157673, + "spread": 0.00134493, + "score": 0.00207241 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00055636, + "spread": 0.001543, + "score": 0.00164024 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00284066, + "spread": 0.00293526, + "score": 0.00408474 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00061851, + "spread": 0.00309986, + "score": 0.00316097 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.03277136, + "spread": 0.04926048, + "score": 0.0591655 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00119263, + "spread": 0.00416776, + "score": 0.00433504 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00046522, + "spread": 0.00176184, + "score": 0.00182223 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00035355, + "spread": 0.0014546, + "score": 0.00149695 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": 0.00324848, + "spread": 0.00304201, + "score": 0.00445044 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.01498092, + "spread": 0.00823609, + "score": 0.01709565 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": 0.01090589, + "spread": 0.03009592, + "score": 0.03201098 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.00261647, + "spread": 0.02028444, + "score": 0.02045249 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.03986131, + "spread": 0.23261264, + "score": 0.23600332 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -0.07560464, + "spread": 0.13700402, + "score": 0.15648055 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00767164, + "spread": 0.04080506, + "score": 0.04151996 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00288865, + "spread": 0.00262754, + "score": 0.0039049 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00347012, + "spread": 0.00113911, + "score": 0.0036523 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -7.837e-05, + "spread": 0.00165124, + "score": 0.0016531 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00056999, + "spread": 0.00122639, + "score": 0.00135237 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -0.00131699, + "spread": 0.00241645, + "score": 0.00275203 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.00090957, + "spread": 0.0203952, + "score": 0.02041547 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00435303, + "spread": 0.00351581, + "score": 0.00559551 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": -0.00704505, + "spread": 0.00572633, + "score": 0.00907874 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": -0.00234884, + "spread": 0.00332176, + "score": 0.00406831 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00185326, + "spread": 0.00632823, + "score": 0.00659401 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00340927, + "spread": 0.00329596, + "score": 0.00474199 + }, + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00018085, + "spread": 0.00287841, + "score": 0.00288409 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00316987, + "spread": 0.00237273, + "score": 0.00395954 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01564405, + "spread": 0.02706897, + "score": 0.03126444 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00234581, + "spread": 0.00916789, + "score": 0.00946325 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00301032, + "spread": 0.01126004, + "score": 0.01165549 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00010875, + "spread": 0.00391333, + "score": 0.00391484 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00460778, + "spread": 0.00965936, + "score": 0.0107021 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.01059292, + "spread": 0.00924604, + "score": 0.01406056 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00909171, + "spread": 0.01093162, + "score": 0.01421828 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00309674, + "spread": 0.00214681, + "score": 0.0037681 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": -0.00143072, + "spread": 0.00363562, + "score": 0.00390701 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00435745, + "spread": 0.00389633, + "score": 0.00584541 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00630039, + "spread": 0.00323323, + "score": 0.00708158 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.06908928, + "spread": 0.03213426, + "score": 0.07619671 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.01449343, + "spread": 0.17802062, + "score": 0.17860963 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 1.28993168, + "spread": 2.21221129, + "score": 2.56082067 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": -2.02811385, + "spread": 3.58814655, + "score": 4.12165519 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.01564106, + "spread": 0.05533218, + "score": 0.05750037 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00188547, + "spread": 0.00929797, + "score": 0.00948721 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00276304, + "spread": 0.00889751, + "score": 0.00931665 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00105854, + "spread": 0.00183345, + "score": 0.00211709 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.04306485, + "spread": 0.08415965, + "score": 0.09453797 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00507191, + "spread": 0.01904063, + "score": 0.01970456 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00663974, + "spread": 0.00513429, + "score": 0.00839327 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 1, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00352009, + "spread": 0.00529956, + "score": 0.00636211 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00126705, + "spread": 0.00166538, + "score": 0.00209258 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01068341, + "spread": 0.02227797, + "score": 0.02470715 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00509417, + "spread": 0.00579142, + "score": 0.00771305 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00436447, + "spread": 0.00496806, + "score": 0.00661289 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00378302, + "spread": 0.00279686, + "score": 0.00470464 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00154083, + "spread": 0.00412255, + "score": 0.00440109 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00510113, + "spread": 0.00664115, + "score": 0.00837415 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": -0.00763642, + "spread": 0.01463878, + "score": 0.01651087 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": -0.00350567, + "spread": 0.00277728, + "score": 0.00447247 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00284194, + "spread": 0.00334086, + "score": 0.00438612 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00472493, + "spread": 0.00449398, + "score": 0.0065208 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00919414, + "spread": 0.00823191, + "score": 0.01234085 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.03212832, + "spread": 0.0870391, + "score": 0.09277949 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": 0.05495255, + "spread": 0.20704202, + "score": 0.2142106 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.13909671, + "spread": 0.20217877, + "score": 0.24540609 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.24222677, + "spread": 0.3647096, + "score": 0.43782063 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.01647143, + "spread": 0.02177919, + "score": 0.02730644 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": -0.00166229, + "spread": 0.00227642, + "score": 0.00281874 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00033015, + "spread": 0.00340086, + "score": 0.00341685 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00173242, + "spread": 0.0030841, + "score": 0.00353737 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.00966759, + "spread": 0.05994207, + "score": 0.06071667 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00071206, + "spread": 0.01380538, + "score": 0.01382373 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00190441, + "spread": 0.00424775, + "score": 0.00465512 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 2, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.00068876, + "spread": 0.00242545, + "score": 0.00252135 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00183128, + "spread": 0.00185698, + "score": 0.00260805 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00670495, + "spread": 0.01465742, + "score": 0.0161182 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00325235, + "spread": 0.00266901, + "score": 0.0042073 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00261509, + "spread": 0.00128774, + "score": 0.00291496 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00029282, + "spread": 0.00309734, + "score": 0.00311115 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00478545, + "spread": 0.00728669, + "score": 0.00871759 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00211698, + "spread": 0.00382002, + "score": 0.00436739 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00225894, + "spread": 0.01066002, + "score": 0.01089673 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.0003161, + "spread": 0.00263877, + "score": 0.00265763 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00404667, + "spread": 0.00278128, + "score": 0.0049103 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00035083, + "spread": 0.00492958, + "score": 0.00494205 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00061215, + "spread": 0.0064336, + "score": 0.00646265 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.0123531, + "spread": 0.03549997, + "score": 0.03758785 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.01901126, + "spread": 0.03850116, + "score": 0.04293911 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": -0.51467847, + "spread": 0.87928925, + "score": 1.0188442 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.10864009, + "spread": 0.1286496, + "score": 0.16838465 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00275083, + "spread": 0.03173845, + "score": 0.03185743 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00089828, + "spread": 0.00061721, + "score": 0.00108988 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00260599, + "spread": 0.00294118, + "score": 0.0039296 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00146263, + "spread": 0.00387517, + "score": 0.00414201 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": -0.00018535, + "spread": 0.00032103, + "score": 0.0003707 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.00786942, + "spread": 0.03997929, + "score": 0.04074643 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": NaN, + "spread": NaN, + "score": NaN + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.00118195, + "spread": 0.01425212, + "score": 0.01430105 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00514818, + "spread": 0.00595623, + "score": 0.00787276 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 3, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.00011576, + "spread": 0.00427549, + "score": 0.00427706 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00152731, + "spread": 0.00157213, + "score": 0.00219186 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00049968, + "spread": 0.00236714, + "score": 0.0024193 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00033225, + "spread": 0.00420217, + "score": 0.00421528 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00279013, + "spread": 0.00361244, + "score": 0.00456449 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 9.956e-05, + "spread": 0.00248941, + "score": 0.0024914 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.0008001, + "spread": 0.00239792, + "score": 0.00252788 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00109923, + "spread": 0.00623894, + "score": 0.00633504 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.06339893, + "spread": 0.01263045, + "score": 0.06464482 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00043278, + "spread": 0.01084267, + "score": 0.01085131 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00090561, + "spread": 0.00162394, + "score": 0.00185938 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00315884, + "spread": 0.00162028, + "score": 0.00355015 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.00176909, + "spread": 0.0026478, + "score": 0.00318442 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": -0.00083703, + "spread": 0.00981064, + "score": 0.00984628 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00027735, + "spread": 0.02718334, + "score": 0.02718475 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.0619254, + "spread": 0.03102378, + "score": 0.06926204 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.09457439, + "spread": 0.31705364, + "score": 0.33085847 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.07663987, + "spread": 0.09112929, + "score": 0.11907232 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": 0.00030753, + "spread": 0.01669991, + "score": 0.01670274 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.0026057, + "spread": 0.00130357, + "score": 0.00291358 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00214592, + "spread": 0.00205528, + "score": 0.00297139 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": 0.00070511, + "spread": 0.00254571, + "score": 0.00264156 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00117214, + "spread": 0.00235787, + "score": 0.00263315 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": 0.00022097, + "spread": 0.00038273, + "score": 0.00044194 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.04421435, + "spread": 0.0501345, + "score": 0.06684592 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.00048858, + "spread": 0.0057116, + "score": 0.00573246 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": -0.00072828, + "spread": 0.00327307, + "score": 0.00335312 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 6, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": 0.0016743, + "spread": 0.00559151, + "score": 0.00583681 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00068629, + "spread": 0.00184793, + "score": 0.00197125 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": -0.00108241, + "spread": 0.00623434, + "score": 0.00632761 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00288707, + "spread": 0.00357957, + "score": 0.00459875 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00186092, + "spread": 0.00134667, + "score": 0.00229707 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00013809, + "spread": 0.00147693, + "score": 0.00148337 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00264107, + "spread": 0.00365638, + "score": 0.00451047 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00026697, + "spread": 0.00353568, + "score": 0.00354575 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.0, 0.05]", + "bias": -0.03751521, + "spread": 0.04689147, + "score": 0.06005165 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.05, 0.1]", + "bias": 0.00211723, + "spread": 0.00376712, + "score": 0.00432132 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.1, 0.15]", + "bias": 0.00126919, + "spread": 0.0015279, + "score": 0.00198629 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.15, 0.2]", + "bias": 0.00048182, + "spread": 0.00154796, + "score": 0.00162122 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.2, 0.25]", + "bias": -0.0005041, + "spread": 0.00328488, + "score": 0.00332333 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.25, 0.3]", + "bias": 0.00262925, + "spread": 0.00694657, + "score": 0.0074275 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.3, 0.35]", + "bias": -0.00948842, + "spread": 0.03334174, + "score": 0.03466557 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.35, 0.4]", + "bias": -0.02164823, + "spread": 0.02098851, + "score": 0.03015234 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.4, 0.45]", + "bias": 0.01713209, + "spread": 0.25572163, + "score": 0.25629487 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": "(0.45, 0.5]", + "bias": 0.50419051, + "spread": 0.57969355, + "score": 0.76827904 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(0.0, 2.0]", + "bias": -0.00590864, + "spread": 0.04226358, + "score": 0.04267461 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(10.0, 12.0]", + "bias": 0.00084424, + "spread": 0.00250798, + "score": 0.00264626 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(12.0, 14.0]", + "bias": 0.00232505, + "spread": 0.00085995, + "score": 0.00247899 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(14.0, 16.0]", + "bias": -0.00022674, + "spread": 0.0014535, + "score": 0.00147108 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(16.0, 18.0]", + "bias": 0.00081464, + "spread": 0.00124414, + "score": 0.00148711 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(18.0, 20.0]", + "bias": -6.589e-05, + "spread": 0.00160329, + "score": 0.00160465 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(2.0, 4.0]", + "bias": -0.02631469, + "spread": 0.02018141, + "score": 0.03316251 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(20.0, 22.0]", + "bias": -0.00065711, + "spread": 0.00204492, + "score": 0.00214791 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(22.0, 24.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(24.0, 26.0]", + "bias": 0.0, + "spread": 0.0, + "score": 0.0 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": -0.001117, + "spread": 0.00713144, + "score": 0.00721839 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.00157068, + "spread": 0.00344089, + "score": 0.00378243 + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(8.0, 10.0]", + "bias": -0.0006748, + "spread": 0.00318421, + "score": 0.00325493 + } + ] + } + } +} diff --git a/benchmarking/baselines/study_toggle_methods_compare.py b/benchmarking/baselines/study_toggle_methods_compare.py new file mode 100644 index 00000000..ec5b3f6a --- /dev/null +++ b/benchmarking/baselines/study_toggle_methods_compare.py @@ -0,0 +1,657 @@ +"""Score the toggle-capable methods on Hill of Towie and diff them against a committed benchmark. + +The regression harness for the **toggle** methods — ``toggle_specialist`` and ``power_model`` — on a +known-stable real dataset. Its job is to answer one question objectively: *did a change to a method +move its numbers, and in which direction?* + +The cases are deliberately small-signal and short: + +- **Profiles:** a placebo (``cp_0pct``) plus a symmetric +/-2% Cp pair. Symmetric magnitudes let a + sign error show up as an asymmetry between the pair, and 2% is the regime a real toggle campaign + actually lives in — far more informative here than a +/-10% signal any method can find. +- **Campaign grid:** 1/2/4/8 **weeks**. A real toggle campaign runs for weeks, and the short end is + where these methods are hardest pressed. + +Each run diffs the fresh bias/spread/score per ``(method, profile, campaign_weeks, condition, bin)`` +against the committed benchmark and logs the deltas plus a per-cell ``benchmark_comparison.csv``. + +**Deltas are raw, and "unchanged" is judged per method** (:data:`_REPRODUCIBILITY`), because the two +methods' reproducibility differs by four orders of magnitude — measured, not assumed. +``toggle_specialist`` is pure arithmetic and reproduces exactly; ``power_model`` does not, despite +its ``seed``, because LightGBM's threaded float reduction order varies (~0.05 pp same-machine). The +max observed delta is logged every run, so a cell just inside its band stays visible. + +**The benchmark is split across files because that reproducibility is also machine-dependent** (F30). +LightGBM's reduction order depends on the machine, so ``power_model`` scores ~0.7 pp against a +benchmark recorded elsewhere — 14x its same-machine noise, and a permanent false MOVED. Its cells +therefore live in a per-platform file (``..._baseline_.json``), while +``toggle_specialist``, which is portable (~5e-07 pp across machines), lives in the shared +``..._baseline_portable.json``. A run diffs the two merged. + +Run from the repo root:: + + uv run python -m benchmarking.baselines.study_toggle_methods_compare + +Restrict to one profile for fast feedback with ``--profiles cp_0pct``. Record with +``--update-baseline`` (deliberately — only when a change is accepted), then commit the JSON(s); it +writes this machine's platform file and, only when they actually change, the portable one. +""" + +from __future__ import annotations + +import argparse +import json +import logging +import os +import platform as platform_module +import subprocess +import sys +from dataclasses import dataclass +from datetime import datetime, timezone +from pathlib import Path +from typing import Any + +import pandas as pd + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TREATMENT_START_RANGE, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, + MIN_PRE_MONTHS, +) +from benchmarking.baselines.example_toggle_study import DEFAULT_TOGGLE_PERIOD +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.power_model import PowerModelMethod +from benchmarking.baselines.toggle_specialist import ToggleSpecialistMethod +from benchmarking.harness import ( + StudyConfig, + conditional_leaderboard, + leaderboard, + plot_campaign_curves, + score_study, +) +from benchmarking.synthetic import HOT_COLUMNS, HOT_RATED_POWER_KW, ConstantCpChange +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# The toggle methods under regression. power_model is the slow one, so it runs last (see +# `on_method_complete` in score_study — order fastest-first for the earliest feedback). +COMPARE_METHODS = ["toggle_specialist", "power_model"] + +# Local to this script rather than added to `overnight_profiles()`: adding profiles there would leave +# study_power_model_compare's committed baseline missing cells (its --update-baseline demands the full +# profile set), disturbing the very benchmark this script exists to protect. +TOGGLE_PROFILES: dict[str, list] = { + "cp_0pct": [ConstantCpChange(delta=0.0)], + "cp_plus_2pct": [ConstantCpChange(delta=0.02)], + "cp_minus_2pct": [ConstantCpChange(delta=-0.02)], +} + +CAMPAIGN_WEEKS = [1, 2, 4, 8] +N_REPLICATES = 4 +SEED = 0 + +_LENGTH_COL = "campaign_weeks" +_DEFAULT_OUTPUT_DIR = Path.home() / "temp" / "wind-up-benchmarking" / "toggle_methods_compare" +_BASELINE_DIR = Path(__file__).resolve().parent +_BASELINE_STEM = "study_toggle_methods_compare_baseline" +# v3 splits the single file into a portable baseline plus one per platform (F30). +_BASELINE_SCHEMA = "toggle_methods_compare_baseline_v3" +# Per-cell metrics recorded and **diffed**. spread/score: lower is better; bias: |bias| nearer 0 is better. +_METRIC_COLS = ["bias", "spread", "score"] +# Recorded per cell but never diffed: wall time is machine- and load-dependent, so diffing it would +# trip the unchanged verdict on every run. Kept because a change that makes a method dramatically +# slower is a regression worth seeing, and because it is the record of what a method actually costs. +_WALL_TIME_COLS = ["wall_time_s_sum", "wall_time_s_mean"] +_CELL_COLS = [*_METRIC_COLS, "mean_estimate", "mean_truth", "n_replicates"] +_MERGE_KEYS = ["method", "profile", _LENGTH_COL, "condition", "condition_bin"] +_PP = 100.0 # fraction -> percentage points + + +@dataclass(frozen=True) +class MethodReproducibility: + """How reproducible a method is, and under what conditions (F30). + + :param band: how close a re-run must land to the benchmark to read "unchanged" (fraction) + :param portable: whether its numbers survive a change of machine, and so whether its cells live + in the shared baseline or in a per-platform one + """ + + band: float + portable: bool + + +# Measured, not assumed. `toggle_specialist` is pure arithmetic: two runs reproduce exactly, and it +# matches a baseline recorded on another machine to ~5e-07 pp, so 1e-7 holds it to an effectively +# bit-exact standard. `power_model` is not reproducible even run to run (LightGBM's threaded float +# reduction order; the seed governs sampling, not that): ~0.05 pp same-machine, but ~0.7 pp against a +# baseline from another machine — hence portable=False. +_REPRODUCIBILITY: dict[str, MethodReproducibility] = { + "toggle_specialist": MethodReproducibility(band=1e-7, portable=True), + "power_model": MethodReproducibility(band=1e-3, portable=False), +} +# An unclassified method is assumed machine-specific: the safe side, since wrongly calling one +# portable produces a permanent, confusing failure on the other machine. +_DEFAULT_REPRODUCIBILITY = MethodReproducibility(band=1e-3, portable=False) + + +def _reproducibility(method: str) -> MethodReproducibility: + """Return ``method``'s reproducibility facts, defaulting to the conservative assumption.""" + return _REPRODUCIBILITY.get(method, _DEFAULT_REPRODUCIBILITY) + + +def _portable_methods() -> set[str]: + """Return the methods whose cells belong in the shared, cross-machine baseline.""" + return {name for name, repro in _REPRODUCIBILITY.items() if repro.portable} + + +def baseline_paths(baseline_dir: Path | None = None, platform: str | None = None) -> tuple[Path, Path]: + """Return ``(portable_path, platform_path)`` for this machine. + + ``sys.platform`` is a proxy for *the machine*, which is only sound while there is one machine per + platform; a second box on the same platform would silently share a file. The recorded fingerprint + (see :func:`_provenance`) is what would expose that. + """ + directory = _BASELINE_DIR if baseline_dir is None else baseline_dir + key = sys.platform if platform is None else platform + return directory / f"{_BASELINE_STEM}_portable.json", directory / f"{_BASELINE_STEM}_{key}.json" + + +def toggle_study() -> StudyConfig: + """Return the study every run scores: toggle mode over the weeks grid.""" + return StudyConfig( + mode="toggle", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=DEFAULT_TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS, + campaign_weeks=CAMPAIGN_WEEKS, + toggle_period=DEFAULT_TOGGLE_PERIOD, + n_replicates=N_REPLICATES, + seed=SEED, + ) + + +def _select_profiles(requested: list[str] | None) -> dict[str, list]: + """Return the profiles to score: all when ``requested`` is ``None``, else the named subset. + + Unknown names fail loudly rather than silently scoring less. + """ + if requested is None: + return TOGGLE_PROFILES + unknown = [name for name in requested if name not in TOGGLE_PROFILES] + if unknown: + msg = f"unknown profile(s) {unknown}; available: {sorted(TOGGLE_PROFILES)}" + raise ValueError(msg) + return {name: TOGGLE_PROFILES[name] for name in requested} + + +def _build_methods(out_dir: Path, *, era5_hourly_df: pd.DataFrame) -> list: + """Construct the HoT-configured toggle methods, fastest first. + + Both report the **power** axis and only that, so the benchmark tracks per-bin behaviour (a change + can leave the headline untouched and still wreck a bin) and so the two methods are compared on a + common axis — ``power`` is the only one ``toggle_specialist`` can offer. ``power_model``'s ws/TI + conditional is deliberately not duplicated here: ``study_power_model_compare`` already tracks it, + and recording it for one method only would make this study's table asymmetric for no gain. + """ + return [ + ToggleSpecialistMethod( + columns=HOT_COLUMNS, + conditions=("power",), + rated_power_kw=HOT_RATED_POWER_KW, + out_dir=out_dir / "toggle_specialist_runs", + ), + PowerModelMethod( + columns=HOT_COLUMNS, + baseline_rated_power_kw=HOT_RATED_POWER_KW, + conditions=("power",), + era5_hourly_df=era5_hourly_df, + out_dir=out_dir / "power_model_runs", + ), + ] + + +def run_study(out_dir: Path, *, profiles: list[str] | None = None) -> pd.DataFrame: + """Score both toggle methods over the profiles, writing a per-profile ``results_*.csv``. + + Returns the concatenated tidy results. + """ + out_dir.mkdir(parents=True, exist_ok=True) + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + context = build_hot_v0_context(wtg_names=DEFAULT_TURBINE_SUBSET) + study = toggle_study() + + all_results = [] + for profile_name, profile in _select_profiles(profiles).items(): + # Per-profile subfolder: a method's run dir is ___ (no profile), so + # profiles sharing a (wtg, window) would otherwise overwrite each other's diagnostics. + methods = _build_methods(out_dir / profile_name, era5_hourly_df=context.reanalysis_datasets[0].data) + logger.info("Scoring profile %s with %s", profile_name, ", ".join(COMPARE_METHODS)) + results = score_study( + scada_df, + profile=profile, + methods=methods, + study=study, + profile_name=profile_name, + ) + results.to_csv(out_dir / f"results_{profile_name}.csv", index=False) + all_results.append(results) + return pd.concat(all_results, ignore_index=True) + + +def _git_commit() -> str: + """Return the short HEAD commit (``-dirty`` if *tracked* files are modified), or ``unknown``. + + ``--untracked-files=no`` is deliberate: only tracked modifications make a run irreproducible from + its commit. An untracked file (a scratch script, an editor artifact, a local CLAUDE.md) has no + bearing on what ``git checkout `` would run, and counting it would make ``--update-baseline`` + unusable for anyone with a stray file in their working copy. + """ + repo = Path(__file__).resolve().parent + try: + commit = subprocess.run( + ["git", "rev-parse", "--short", "HEAD"], # noqa: S607 + cwd=repo, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + dirty = subprocess.run( + ["git", "status", "--porcelain", "--untracked-files=no"], # noqa: S607 + cwd=repo, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + except (subprocess.SubprocessError, OSError): + return "unknown" + return f"{commit}-dirty" if dirty else commit + + +def methods_leaderboard(results: pd.DataFrame) -> pd.DataFrame: + """One row per (method, profile, campaign_weeks, condition, bin) of bias/spread/score. + + Both the headline (``condition == "overall"``) and the per-power-bin cells are recorded. Tracking + only the headline would miss the failure mode this study most needs to catch: a change can leave + the overall number untouched and still wreck an individual bin. + """ + overall = leaderboard(results, length_col=_LENGTH_COL).assign(condition="overall", condition_bin="overall") + conditional = conditional_leaderboard(results, length_col=_LENGTH_COL) + stacked = pd.concat([overall[[*_MERGE_KEYS, *_CELL_COLS]], conditional[[*_MERGE_KEYS, *_CELL_COLS]]]) + # Wall time is per *estimate*, so only the headline rows carry it (a per-bin row has no fit of its + # own and gets NaN). It is recorded but never diffed — see _METRIC_COLS — because it is machine- + # and load-dependent; it exists so a change that makes a method dramatically slower is visible. + stacked = stacked.merge(overall[[*_MERGE_KEYS, *_WALL_TIME_COLS]], on=_MERGE_KEYS, how="left") + return stacked.sort_values(_MERGE_KEYS).reset_index(drop=True) + + +def _provenance(study: StudyConfig, lb: pd.DataFrame, *, git_commit: str) -> dict[str, Any]: + """Return the context needed to interpret a recorded baseline, including which machine made it. + + The machine fingerprint exists because a ``power_model`` MOVED against a baseline from another + machine is expected rather than a regression, and without this the file cannot say so (F30). + """ + return { + "schema": _BASELINE_SCHEMA, + "recorded_utc": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"), + "git_commit": git_commit, + "platform": sys.platform, + "cpu_count": os.cpu_count(), + "python_version": platform_module.python_version(), + "lightgbm_version": _lightgbm_version(), + "n_replicates": study.n_replicates, + "seed": study.seed, + "campaign_weeks": list(study.campaign_lengths), + "profiles": sorted(lb["profile"].unique()), + } + + +def _lightgbm_version() -> str | None: + """LightGBM's version, or ``None`` when it is not installed (it is an optional dependency).""" + try: + import lightgbm # noqa: PLC0415 + except ImportError: + return None + return str(lightgbm.__version__) + + +def _write_baseline(path: Path, *, cells: pd.DataFrame, provenance: dict[str, Any]) -> None: + """Write one baseline file: its cells plus provenance. + + A no-op when there are no cells, rather than writing an empty file: a run that scored none of this + file's methods knows nothing about them, and overwriting a good baseline with zero cells would + silently destroy it. + """ + if cells.empty: + logger.info("No cells for %s in this run — leaving it alone.", path.name) + return + doc = {**provenance, "methods": sorted(cells["method"].unique()), "cells": cells.round(8).to_dict(orient="records")} + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(doc, indent=2) + "\n") + logger.info("Recorded %s (%d cells, methods %s).", path.name, len(cells), doc["methods"]) + + +def record_baselines( + lb: pd.DataFrame, *, study: StudyConfig, git_commit: str, baseline_dir: Path | None = None +) -> None: + """Record this machine's baselines: the portable cells (shared) and the rest (per platform). + + ``git_commit`` is captured *before* the run, so a commit landing mid-sweep cannot stamp the + baseline with code that never produced it. + + The portable file is only rewritten when it has to be. Its cells are the same on every machine by + definition, so an unchanged recording leaves the file — and its provenance — untouched, which + keeps the two laptops from fighting over it. Its ``git_commit`` therefore records when those + numbers were last *established*, not who last ran a recording. + """ + portable_path, platform_path = baseline_paths(baseline_dir) + provenance = _provenance(study, lb, git_commit=git_commit) + portable_names = _portable_methods() + portable = lb[lb["method"].isin(portable_names)] + machine_specific = lb[~lb["method"].isin(portable_names)] + + _check_portable_or_raise(portable, path=portable_path, git_commit=git_commit) + if _portable_cells_match(portable, path=portable_path): + logger.info("Portable baseline unchanged at %s — portability confirmed, not rewritten.", portable_path) + else: + _write_baseline(portable_path, cells=portable, provenance=provenance) + _write_baseline(platform_path, cells=machine_specific, provenance=provenance) + logger.info("Recorded benchmark(s) at commit %s on %s. Commit the JSON(s).", git_commit, sys.platform) + + +def _portable_cells_match(portable: pd.DataFrame, *, path: Path) -> bool: + """Whether the fresh portable cells match the committed ones **within each method's band**. + + Not a bit-exact comparison, for two reasons. Portability is a claim at the band's precision + (measured ~5e-07 pp), and `round(8)`'s 1e-8 resolution is close enough to that to flip a last + digit. And the recorded cells carry wall time, which differs every run by construction — an exact + comparison would rewrite the shared file on every recording and hand the two laptops a conflict. + """ + loaded = _load_baseline(path) + if loaded is None: + return False + base, _ = loaded + merged = portable.merge(base, on=_MERGE_KEYS, how="outer", suffixes=("", "_base"), indicator=True) + if (merged["_merge"] != "both").any(): + return False # a cell appeared or vanished: not the same set of numbers + for method, group in merged.groupby("method"): + band = _reproducibility(str(method)).band + for col in _METRIC_COLS: + fresh, old = group[col], group[f"{col}_base"] + delta = (fresh - old).abs() + both_nan = fresh.isna() & old.isna() # an empty bin is NaN in both and matches + if ((delta > band) | (~both_nan & delta.isna())).any(): + return False + return True + + +def _check_portable_or_raise(portable: pd.DataFrame, *, path: Path, git_commit: str) -> None: + """Refuse to record when a portable method's cells moved **at the same commit**. + + From one machine "the numbers moved" is ambiguous: it means either the method changed (which is + what --update-baseline is for) or portability broke. The commit disambiguates — same code + producing different numbers on a different machine is a portability break, and since + ``--update-baseline`` refuses a dirty tree the commit is trustworthy enough to lean on. + """ + loaded = _load_baseline(path) + if loaded is None or portable.empty: + return + _, prov = loaded + if prov.get("git_commit") != git_commit or _portable_cells_match(portable, path=path): + return + msg = ( + f"portable baseline {path.name} was recorded at this same commit ({git_commit}) on " + f"{prov.get('platform')}, but this machine ({sys.platform}) produces different cells for " + f"{sorted(portable['method'].unique())}. Same code, different numbers, different machine: either a " + f"method marked portable=True is not (check _REPRODUCIBILITY), or that file's commit is wrong. " + f"Refusing to overwrite — this is the check the portable/per-platform split exists to make." + ) + raise ValueError(msg) + + +def _load_baseline(path: Path) -> tuple[pd.DataFrame, dict[str, Any]] | None: + """Load one baseline's cells (+ provenance), or ``None`` if absent/stale.""" + if not path.exists(): + return None + doc = json.loads(path.read_text()) + if doc.get("schema") != _BASELINE_SCHEMA: + logger.warning( + "Baseline %s has schema %r, expected %r — run --update-baseline to regenerate.", + path, + doc.get("schema"), + _BASELINE_SCHEMA, + ) + return None + return pd.DataFrame(doc["cells"]), doc + + +def _warn_on_fingerprint_mismatch(prov: dict[str, Any], *, path: Path) -> None: + """Warn when the **platform** baseline was recorded somewhere unlike this machine. Never fatal. + + Only meaningful for the platform file, whose methods are machine-specific by definition. The + portable file is *expected* to come from the other laptop — that is the point of it — so warning + there would be noise on every run. Fields recorded as ``None`` (unrecoverable when the v2 file was + migrated) make no claim and are skipped. + """ + current = {"platform": sys.platform, "cpu_count": os.cpu_count(), "lightgbm_version": _lightgbm_version()} + differing = {k: (prov.get(k), v) for k, v in current.items() if prov.get(k) is not None and prov.get(k) != v} + if differing: + detail = ", ".join(f"{k}: recorded {was!r}, now {now!r}" for k, (was, now) in differing.items()) + logger.warning( + "%s holds machine-specific cells but was recorded on a machine unlike this one (%s). A MOVED " + "verdict may be that rather than your change — see F30.", + path.name, + detail, + ) + + +def load_merged_baseline(baseline_dir: Path | None = None) -> tuple[pd.DataFrame, dict[str, Any]] | None: + """Load the portable + this-platform baselines merged, or ``None`` when neither is recorded. + + Either half missing is a warning, not an error: a fresh machine with no platform file still gets + its portable regression check for free. + """ + portable_path, platform_path = baseline_paths(baseline_dir) + frames, provenance = [], {} + for path in (portable_path, platform_path): + loaded = _load_baseline(path) + if loaded is None: + logger.warning( + "No usable benchmark at %s — %s. Run --update-baseline on this machine to record it.", + path.name, + "portable cells will not be diffed" + if path == portable_path + else f"{sys.platform} cells will not be diffed", + ) + continue + cells, prov = loaded + if path == platform_path: # the portable file is meant to come from the other machine + _warn_on_fingerprint_mismatch(prov, path=path) + frames.append(cells) + provenance[path.name] = prov + if not frames: + return None + return pd.concat(frames, ignore_index=True), provenance + + +def compare_to_benchmark(lb: pd.DataFrame, *, comparison_dir: Path, baseline_dir: Path | None = None) -> pd.DataFrame: + """Diff the fresh cells against the committed benchmarks; write the per-cell CSV and log the deltas. + + Returns the merged frame (empty if nothing is recorded yet). Deltas are raw; the unchanged/moved + verdict applies each method's band from :data:`_REPRODUCIBILITY`. + """ + loaded = load_merged_baseline(baseline_dir) + if loaded is None: + logger.warning("No benchmark recorded for this machine yet — run with --update-baseline to set it.") + return pd.DataFrame() + base, prov = loaded + base = base[base["profile"].isin(lb["profile"].unique())] # scope to the profiles actually run + merged = lb.merge(base, on=_MERGE_KEYS, how="outer", suffixes=("", "_base")) + for col in _METRIC_COLS: + merged[f"d_{col}"] = merged[col] - merged[f"{col}_base"] + merged = merged.sort_values(_MERGE_KEYS) + comparison_dir.mkdir(parents=True, exist_ok=True) + merged.to_csv(comparison_dir / "benchmark_comparison.csv", index=False) + + show = pd.DataFrame( + { + "method": merged["method"], + "profile": merged["profile"], + _LENGTH_COL: merged[_LENGTH_COL], + "bias": merged["bias"] * _PP, + "d_bias": merged["d_bias"] * _PP, + "spread": merged["spread"] * _PP, + "d_spread": merged["d_spread"] * _PP, + "score": merged["score"] * _PP, + "d_score": merged["d_score"] * _PP, + } + ).round(6) + logger.info( + "Toggle methods vs benchmark (%s) [pp]; unchanged = within each method's band (%s):\n%s", + "; ".join( + f"{name} recorded {p.get('recorded_utc', '?')} at {p.get('git_commit', '?')}" for name, p in prov.items() + ), + ", ".join(f"{m} {r.band * _PP:g} pp" for m, r in _REPRODUCIBILITY.items()), + show.to_string(index=False), + ) + _log_unchanged_verdict(merged) + return merged + + +def _log_unchanged_verdict(merged: pd.DataFrame, *, atol: dict[str, float] | None = None) -> None: + """Log, per method, whether every diffed cell matches the benchmark within that method's band. + + The max observed delta is always reported, so "unchanged" never hides a cell sitting just inside + its band. + """ + delta_cols = [f"d_{col}" for col in _METRIC_COLS] + for method, group in merged.groupby("method"): + comparable = group.dropna(subset=delta_cols) + if comparable.empty: + logger.info("%s: no cells line up with the benchmark (new method or new cells).", method) + continue + band = ( + _reproducibility(str(method)).band if atol is None else atol.get(str(method), _DEFAULT_REPRODUCIBILITY.band) + ) + worst = float(comparable[delta_cols].abs().to_numpy().max()) + moved = comparable[(comparable[delta_cols].abs() > band).any(axis=1)] + if moved.empty: + logger.info( + "%s: UNCHANGED — all %d cells within +/-%.3g pp of the benchmark (max delta %.3g pp).", + method, + len(comparable), + band * _PP, + worst * _PP, + ) + else: + logger.warning( + "%s: MOVED — %d of %d cells differ by more than +/-%.3g pp (max delta %.3g pp):\n%s", + method, + len(moved), + len(comparable), + band * _PP, + worst * _PP, + moved[[*_MERGE_KEYS, *delta_cols]].to_string(index=False), + ) + + +def plot_results(lb: pd.DataFrame, comparison_dir: Path) -> None: + """Write one campaign-length curve per profile (both methods overlaid), from the headline rows. + + Restricted to ``condition == "overall"``: the leaderboard now also carries per-bin rows, and a + campaign curve drawn over both would silently average the headline together with six power bins. + """ + comparison_dir.mkdir(parents=True, exist_ok=True) + headline = lb[lb["condition"] == "overall"] + for profile in sorted(headline["profile"].unique()): + plot_campaign_curves( + headline[headline["profile"] == profile], + save_path=comparison_dir / f"campaign_curves_{profile}.png", + title=f"toggle - {profile} (toggle_specialist vs power_model)", + length_col=_LENGTH_COL, + ) + + +def main() -> None: + """Score the toggle methods over the profiles, then diff against (or record) the benchmark.""" + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument( + "--output-dir", + type=Path, + default=_DEFAULT_OUTPUT_DIR, + help="where method runs, results and the comparison are written", + ) + parser.add_argument( + "--profiles", + nargs="+", + choices=sorted(TOGGLE_PROFILES), + default=None, + help="restrict to a subset of profiles for fast feedback (default: all three). " + "Cannot be combined with --update-baseline.", + ) + parser.add_argument( + "--baseline-dir", + type=Path, + default=None, + help="directory holding the committed benchmark JSONs (default: next to this script)", + ) + parser.add_argument( + "--update-baseline", + action="store_true", + help="overwrite this machine's recorded benchmark with this run (do this deliberately, only when " + "a change is accepted); without it, the run is diffed against the benchmark", + ) + args = parser.parse_args() + if args.profiles is not None and args.update_baseline: + # A recording rewrites the cells wholesale, so a subset run would drop the other profiles from + # the committed benchmark. Refuse rather than silently corrupt it. + parser.error("--update-baseline needs the full profile set; do not combine it with --profiles") + + # Captured *before* the sweep: the run takes ~15 min, so reading HEAD afterwards would stamp the + # baseline with whatever was committed meanwhile rather than the code that actually ran. + git_commit = _git_commit() + if args.update_baseline and git_commit.endswith("-dirty"): + # The committed benchmark is only worth anything if a reader can check out the commit and + # reproduce it. Recording from a dirty tree bakes in changes that commit does not contain, so + # refuse rather than write an untraceable baseline. Commit first, then record, then commit the + # JSON. (Without --update-baseline a dirty tree is fine — that run only reports.) + parser.error( + f"refusing to --update-baseline from a dirty tree (commit {git_commit}): the committed " + f"benchmark must be reproducible from its commit. Commit your changes first, then re-run." + ) + + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s", force=True) + output_dir = args.output_dir.expanduser() + baseline_dir = args.baseline_dir.expanduser() if args.baseline_dir is not None else None + + results = run_study(output_dir, profiles=args.profiles) + lb = methods_leaderboard(results) + lb.to_csv(output_dir / "leaderboard.csv", index=False) + # The headline goes to the log; the per-bin rows are ~6x more numerous and would bury it. They are + # all in leaderboard.csv, and the benchmark diff reports any that move. + headline = lb[lb["condition"] == "overall"] + logger.info( + "Leaderboard (headline; %d per-bin rows also recorded, see leaderboard.csv):\n%s", + len(lb) - len(headline), + headline[["method", "profile", _LENGTH_COL, *_METRIC_COLS]].to_string(index=False), + ) + + comparison_dir = output_dir / "comparison" + plot_results(lb, comparison_dir) + if args.update_baseline: + record_baselines(lb, study=toggle_study(), git_commit=git_commit, baseline_dir=baseline_dir) + else: + compare_to_benchmark(lb, comparison_dir=comparison_dir, baseline_dir=baseline_dir) + logger.info("All done. Outputs under %s", output_dir) + + +if __name__ == "__main__": + main() diff --git a/benchmarking/baselines/study_toggle_methods_compare_baseline_linux.json b/benchmarking/baselines/study_toggle_methods_compare_baseline_linux.json new file mode 100644 index 00000000..935c4606 --- /dev/null +++ b/benchmarking/baselines/study_toggle_methods_compare_baseline_linux.json @@ -0,0 +1,1287 @@ +{ + "schema": "toggle_methods_compare_baseline_v3", + "recorded_utc": "2026-07-15T20:22:16Z", + "git_commit": "8ae041a", + "platform": "linux", + "cpu_count": 12, + "python_version": "3.13.7", + "lightgbm_version": "4.6.0", + "n_replicates": 4, + "seed": 0, + "campaign_weeks": [ + 1, + 2, + 4, + 8 + ], + "profiles": [ + "cp_0pct", + "cp_minus_2pct", + "cp_plus_2pct" + ], + "methods": [ + "power_model" + ], + "cells": [ + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00214143, + "spread": 0.00494805, + "score": 0.00539156, + "mean_estimate": 0.00214143, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 64.0599861, + "wall_time_s_mean": 16.01499653 + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.03577532, + "spread": 0.03243449, + "score": 0.04828944, + "mean_estimate": 0.03577532, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00372399, + "spread": 0.00510886, + "score": 0.00632207, + "mean_estimate": -0.00372399, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00062516, + "spread": 0.00094793, + "score": 0.00113552, + "mean_estimate": -0.00062516, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00132153, + "spread": 0.00348558, + "score": 0.0037277, + "mean_estimate": -0.00066077, + "mean_truth": 0.0, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00445521, + "spread": 0.00315209, + "score": 0.00545753, + "mean_estimate": 0.00445521, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00323203, + "spread": 0.00643125, + "score": 0.00719771, + "mean_estimate": -0.00323203, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.0014608, + "spread": 0.00100989, + "score": 0.0017759, + "mean_estimate": -0.0014608, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 67.61684672, + "wall_time_s_mean": 16.90421168 + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04335944, + "spread": 0.06327591, + "score": 0.07670647, + "mean_estimate": 0.04335944, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00021099, + "spread": 0.01508551, + "score": 0.01508699, + "mean_estimate": 0.00021099, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00588797, + "spread": 0.0078592, + "score": 0.00982015, + "mean_estimate": -0.00588797, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00167823, + "spread": 0.00169212, + "score": 0.00238322, + "mean_estimate": -0.00167823, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00101257, + "spread": 0.01438763, + "score": 0.01442322, + "mean_estimate": -0.00101257, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00300964, + "spread": 0.00628776, + "score": 0.00697093, + "mean_estimate": -0.00300964, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00274203, + "spread": 0.00216862, + "score": 0.00349595, + "mean_estimate": -0.00274203, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 72.61110581, + "wall_time_s_mean": 18.15277645 + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0254561, + "spread": 0.03572977, + "score": 0.0438706, + "mean_estimate": 0.0254561, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00168801, + "spread": 0.01316243, + "score": 0.01327023, + "mean_estimate": 0.00168801, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00099004, + "spread": 0.01288916, + "score": 0.01292713, + "mean_estimate": -0.00099004, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00010917, + "spread": 0.00431323, + "score": 0.00431461, + "mean_estimate": -0.00010917, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.0023455, + "spread": 0.01363268, + "score": 0.01383297, + "mean_estimate": -0.0023455, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00900789, + "spread": 0.0078359, + "score": 0.01193915, + "mean_estimate": -0.00900789, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00065406, + "spread": 0.00173836, + "score": 0.00185734, + "mean_estimate": -0.00065406, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 84.40903572, + "wall_time_s_mean": 21.10225893 + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01218539, + "spread": 0.02035789, + "score": 0.02372609, + "mean_estimate": 0.01218539, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00151874, + "spread": 0.0079668, + "score": 0.00811027, + "mean_estimate": -0.00151874, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00161252, + "spread": 0.00346864, + "score": 0.00382514, + "mean_estimate": 0.00161252, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00394967, + "spread": 0.00356513, + "score": 0.00532072, + "mean_estimate": -0.00394967, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00278576, + "spread": 0.00300627, + "score": 0.00409855, + "mean_estimate": 0.00278576, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00240404, + "spread": 0.00188757, + "score": 0.00305652, + "mean_estimate": -0.00240404, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00218439, + "spread": 0.00489418, + "score": 0.00535953, + "mean_estimate": -0.01415793, + "mean_truth": -0.01634232, + "n_replicates": 4, + "wall_time_s_sum": 59.87349133, + "wall_time_s_mean": 14.96837283 + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00339514, + "spread": 0.05095515, + "score": 0.05106813, + "mean_estimate": -0.03014664, + "mean_truth": -0.03354178, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00551095, + "spread": 0.01487311, + "score": 0.01586127, + "mean_estimate": -0.01406352, + "mean_truth": -0.01957448, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01031093, + "spread": 0.00606285, + "score": 0.01196133, + "mean_estimate": -0.00314363, + "mean_truth": -0.01345456, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00309268, + "spread": 0.00372115, + "score": 0.00483856, + "mean_estimate": -0.00178903, + "mean_truth": -0.00048538, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.0002332, + "spread": 0.01830548, + "score": 0.01830696, + "mean_estimate": -0.0197662, + "mean_truth": -0.01999941, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00722447, + "spread": 0.01353728, + "score": 0.01534441, + "mean_estimate": -0.01275874, + "mean_truth": -0.01998321, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00141212, + "spread": 0.00098232, + "score": 0.00172019, + "mean_estimate": -0.01607685, + "mean_truth": -0.01466473, + "n_replicates": 4, + "wall_time_s_sum": 58.63149056, + "wall_time_s_mean": 14.65787264 + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04076985, + "spread": 0.06483753, + "score": 0.07659038, + "mean_estimate": 0.018606, + "mean_truth": -0.02216385, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00108184, + "spread": 0.01556498, + "score": 0.01560253, + "mean_estimate": -0.02060488, + "mean_truth": -0.01952304, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0008808, + "spread": 0.01162295, + "score": 0.01165628, + "mean_estimate": -0.01096174, + "mean_truth": -0.01184254, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00142757, + "spread": 0.00206314, + "score": 0.00250888, + "mean_estimate": -0.00197217, + "mean_truth": -0.00054461, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00218442, + "spread": 0.01304436, + "score": 0.01322599, + "mean_estimate": -0.02218384, + "mean_truth": -0.01999942, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00393965, + "spread": 0.00540045, + "score": 0.00668474, + "mean_estimate": -0.02392227, + "mean_truth": -0.01998261, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00269244, + "spread": 0.00212619, + "score": 0.00343072, + "mean_estimate": -0.01706125, + "mean_truth": -0.01436881, + "n_replicates": 4, + "wall_time_s_sum": 63.68285428, + "wall_time_s_mean": 15.92071357 + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02593602, + "spread": 0.03372401, + "score": 0.04254394, + "mean_estimate": 0.00444479, + "mean_truth": -0.02149123, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00325386, + "spread": 0.01292994, + "score": 0.01333307, + "mean_estimate": -0.0162613, + "mean_truth": -0.01951516, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00057341, + "spread": 0.01322971, + "score": 0.01324213, + "mean_estimate": -0.01064005, + "mean_truth": -0.01121346, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00057519, + "spread": 0.00427197, + "score": 0.00431052, + "mean_estimate": -0.00119341, + "mean_truth": -0.00061822, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00305027, + "spread": 0.01214221, + "score": 0.01251948, + "mean_estimate": -0.02304971, + "mean_truth": -0.01999945, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00940894, + "spread": 0.00780227, + "score": 0.01222308, + "mean_estimate": -0.02939201, + "mean_truth": -0.01998307, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00063105, + "spread": 0.00171127, + "score": 0.00182392, + "mean_estimate": -0.01475207, + "mean_truth": -0.01412102, + "n_replicates": 4, + "wall_time_s_sum": 76.15675167, + "wall_time_s_mean": 19.03918792 + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01047913, + "spread": 0.01860025, + "score": 0.02134904, + "mean_estimate": -0.01081397, + "mean_truth": -0.02129311, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00057428, + "spread": 0.00639227, + "score": 0.00641802, + "mean_estimate": -0.01894376, + "mean_truth": -0.01951804, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00257836, + "spread": 0.00255889, + "score": 0.00363261, + "mean_estimate": -0.00864315, + "mean_truth": -0.0112215, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00437483, + "spread": 0.00415485, + "score": 0.0060334, + "mean_estimate": -0.0049394, + "mean_truth": -0.00056457, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00342832, + "spread": 0.00344346, + "score": 0.00485909, + "mean_estimate": -0.01657115, + "mean_truth": -0.01999947, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00370887, + "spread": 0.00138652, + "score": 0.00395957, + "mean_estimate": -0.02369239, + "mean_truth": -0.01998352, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00210427, + "spread": 0.00501263, + "score": 0.00543639, + "mean_estimate": 0.01844659, + "mean_truth": 0.01634232, + "n_replicates": 4, + "wall_time_s_sum": 66.57034101, + "wall_time_s_mean": 16.64258525 + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02988906, + "spread": 0.03193474, + "score": 0.04373995, + "mean_estimate": 0.06343084, + "mean_truth": 0.03354178, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.01373532, + "spread": 0.00628342, + "score": 0.01510432, + "mean_estimate": 0.00583916, + "mean_truth": 0.01957448, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.01144839, + "spread": 0.00491394, + "score": 0.01245843, + "mean_estimate": 0.00200617, + "mean_truth": 0.01345456, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00084833, + "spread": 0.00317659, + "score": 0.00328792, + "mean_estimate": -0.00018147, + "mean_truth": 0.00048538, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.02759456, + "spread": 0.02040003, + "score": 0.03431648, + "mean_estimate": 0.04759397, + "mean_truth": 0.01999941, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.01185033, + "spread": 0.00937214, + "score": 0.01510851, + "mean_estimate": 0.00813289, + "mean_truth": 0.01998321, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00150428, + "spread": 0.0010294, + "score": 0.00182278, + "mean_estimate": 0.01316044, + "mean_truth": 0.01466473, + "n_replicates": 4, + "wall_time_s_sum": 62.75173502, + "wall_time_s_mean": 15.68793375 + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04369216, + "spread": 0.06478134, + "score": 0.07813851, + "mean_estimate": 0.06585601, + "mean_truth": 0.02216385, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00197185, + "spread": 0.01717944, + "score": 0.01729224, + "mean_estimate": 0.02149489, + "mean_truth": 0.01952304, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.01204998, + "spread": 0.0050029, + "score": 0.01304726, + "mean_estimate": -0.00020744, + "mean_truth": 0.01184254, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00228366, + "spread": 0.00187632, + "score": 0.00295562, + "mean_estimate": -0.00173906, + "mean_truth": 0.00054461, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.0007196, + "spread": 0.01320906, + "score": 0.01322864, + "mean_estimate": 0.02071903, + "mean_truth": 0.01999942, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00277078, + "spread": 0.00987262, + "score": 0.01025406, + "mean_estimate": 0.01721183, + "mean_truth": 0.01998261, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00279368, + "spread": 0.00220765, + "score": 0.00356067, + "mean_estimate": 0.01157513, + "mean_truth": 0.01436881, + "n_replicates": 4, + "wall_time_s_sum": 70.805462, + "wall_time_s_mean": 17.7013655 + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02671615, + "spread": 0.03589581, + "score": 0.04474663, + "mean_estimate": 0.04820738, + "mean_truth": 0.02149123, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 1.508e-05, + "spread": 0.0135403, + "score": 0.01354031, + "mean_estimate": 0.01953024, + "mean_truth": 0.01951516, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.0015262, + "spread": 0.01176933, + "score": 0.01186788, + "mean_estimate": 0.00968726, + "mean_truth": 0.01121346, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.0001978, + "spread": 0.00461497, + "score": 0.00461921, + "mean_estimate": 0.00042042, + "mean_truth": 0.00061822, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.0011679, + "spread": 0.01257586, + "score": 0.01262997, + "mean_estimate": 0.01883154, + "mean_truth": 0.01999945, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00941784, + "spread": 0.00788807, + "score": 0.01228484, + "mean_estimate": 0.01056523, + "mean_truth": 0.01998307, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00067745, + "spread": 0.00176554, + "score": 0.00189105, + "mean_estimate": 0.01344357, + "mean_truth": 0.01412102, + "n_replicates": 4, + "wall_time_s_sum": 80.52949864, + "wall_time_s_mean": 20.13237466 + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00990576, + "spread": 0.01900177, + "score": 0.02142875, + "mean_estimate": 0.03119887, + "mean_truth": 0.02129311, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00167725, + "spread": 0.00813979, + "score": 0.00831079, + "mean_estimate": 0.01784079, + "mean_truth": 0.01951804, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00079507, + "spread": 0.00265704, + "score": 0.00277345, + "mean_estimate": 0.01201658, + "mean_truth": 0.0112215, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00383931, + "spread": 0.004025, + "score": 0.00556246, + "mean_estimate": -0.00327475, + "mean_truth": 0.00056457, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00345705, + "spread": 0.00359088, + "score": 0.00498454, + "mean_estimate": 0.02345652, + "mean_truth": 0.01999947, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00329187, + "spread": 0.00269912, + "score": 0.00425695, + "mean_estimate": 0.01669166, + "mean_truth": 0.01998352, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + } + ] +} diff --git a/benchmarking/baselines/study_toggle_methods_compare_baseline_portable.json b/benchmarking/baselines/study_toggle_methods_compare_baseline_portable.json new file mode 100644 index 00000000..6a29aa6a --- /dev/null +++ b/benchmarking/baselines/study_toggle_methods_compare_baseline_portable.json @@ -0,0 +1,1287 @@ +{ + "schema": "toggle_methods_compare_baseline_v3", + "recorded_utc": "2026-07-15T13:10:41Z", + "git_commit": "de85f84", + "platform": "win32", + "cpu_count": null, + "python_version": null, + "lightgbm_version": null, + "n_replicates": 4, + "seed": 0, + "campaign_weeks": [ + 1, + 2, + 4, + 8 + ], + "profiles": [ + "cp_0pct", + "cp_minus_2pct", + "cp_plus_2pct" + ], + "methods": [ + "toggle_specialist" + ], + "cells": [ + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00689431, + "spread": 0.01011119, + "score": 0.01223796, + "mean_estimate": -0.00689431, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 0.1919447, + "wall_time_s_mean": 0.04798618 + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02642315, + "spread": 0.05665577, + "score": 0.06251447, + "mean_estimate": 0.02642315, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.01309939, + "spread": 0.02572648, + "score": 0.02886946, + "mean_estimate": 0.01309939, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.0091105, + "spread": 0.02859426, + "score": 0.03001054, + "mean_estimate": -0.0091105, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.01213439, + "spread": 0.01494317, + "score": 0.01924946, + "mean_estimate": -0.01213439, + "mean_truth": 0.0, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.0119027, + "spread": 0.01076691, + "score": 0.01604994, + "mean_estimate": -0.0119027, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.01613312, + "spread": 0.04407323, + "score": 0.04693322, + "mean_estimate": -0.01613312, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00066874, + "spread": 0.00082603, + "score": 0.0010628, + "mean_estimate": 0.00066874, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 0.2210497, + "wall_time_s_mean": 0.05526242 + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04882187, + "spread": 0.07102599, + "score": 0.08618739, + "mean_estimate": 0.04882187, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.01355646, + "spread": 0.01125439, + "score": 0.01761928, + "mean_estimate": 0.01355646, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00817509, + "spread": 0.0275924, + "score": 0.02877799, + "mean_estimate": 0.00817509, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00462446, + "spread": 0.00734518, + "score": 0.0086797, + "mean_estimate": -0.00462446, + "mean_truth": 0.0, + "n_replicates": 3, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00283014, + "spread": 0.01806311, + "score": 0.01828348, + "mean_estimate": 0.00283014, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.0060602, + "spread": 0.02163549, + "score": 0.02246821, + "mean_estimate": -0.0060602, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00099344, + "spread": 0.00271515, + "score": 0.00289118, + "mean_estimate": -0.00099344, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 0.2311662, + "wall_time_s_mean": 0.05779155 + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02504334, + "spread": 0.04183913, + "score": 0.04876148, + "mean_estimate": 0.02504334, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00318488, + "spread": 0.01077895, + "score": 0.01123963, + "mean_estimate": -0.00318488, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01017915, + "spread": 0.02394131, + "score": 0.02601541, + "mean_estimate": 0.01017915, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.0187016, + "spread": 0.0294226, + "score": 0.03486315, + "mean_estimate": 0.0187016, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00176062, + "spread": 0.01386017, + "score": 0.01397154, + "mean_estimate": -0.00176062, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00450401, + "spread": 0.01636677, + "score": 0.0169752, + "mean_estimate": -0.00450401, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00033735, + "spread": 0.00359861, + "score": 0.00361438, + "mean_estimate": 0.00033735, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 0.351947, + "wall_time_s_mean": 0.08798675 + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0074282, + "spread": 0.01814001, + "score": 0.01960199, + "mean_estimate": 0.0074282, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00140765, + "spread": 0.0081515, + "score": 0.00827215, + "mean_estimate": 0.00140765, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00651053, + "spread": 0.0092801, + "score": 0.01133611, + "mean_estimate": 0.00651053, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00367226, + "spread": 0.00184322, + "score": 0.00410888, + "mean_estimate": 0.00367226, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00164199, + "spread": 0.00890525, + "score": 0.00905536, + "mean_estimate": 0.00164199, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00625946, + "spread": 0.0126162, + "score": 0.01408365, + "mean_estimate": -0.00625946, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00714659, + "spread": 0.00931281, + "score": 0.01173892, + "mean_estimate": -0.02348891, + "mean_truth": -0.01634232, + "n_replicates": 4, + "wall_time_s_sum": 0.2014748, + "wall_time_s_mean": 0.0503687 + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0376895, + "spread": 0.07028972, + "score": 0.07975677, + "mean_estimate": 0.00414772, + "mean_truth": -0.03354178, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.01334063, + "spread": 0.02546397, + "score": 0.02874693, + "mean_estimate": -0.00623385, + "mean_truth": -0.01957448, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00819249, + "spread": 0.02937751, + "score": 0.03049844, + "mean_estimate": -0.02164705, + "mean_truth": -0.01345456, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.01307733, + "spread": 0.01611614, + "score": 0.02075443, + "mean_estimate": -0.01356271, + "mean_truth": -0.00048538, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.01166412, + "spread": 0.01055144, + "score": 0.01572846, + "mean_estimate": -0.03166353, + "mean_truth": -0.01999941, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.01577672, + "spread": 0.04321105, + "score": 0.04600109, + "mean_estimate": -0.03575994, + "mean_truth": -0.01998321, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00059889, + "spread": 0.00174801, + "score": 0.00184776, + "mean_estimate": -0.01526362, + "mean_truth": -0.01466473, + "n_replicates": 4, + "wall_time_s_sum": 0.2217921, + "wall_time_s_mean": 0.05544803 + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04899736, + "spread": 0.07030662, + "score": 0.08569575, + "mean_estimate": 0.02683351, + "mean_truth": -0.02216385, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.01432673, + "spread": 0.01123463, + "score": 0.01820638, + "mean_estimate": -0.0051963, + "mean_truth": -0.01952304, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00833484, + "spread": 0.02550988, + "score": 0.02683698, + "mean_estimate": -0.0035077, + "mean_truth": -0.01184254, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00538071, + "spread": 0.00789269, + "score": 0.00955231, + "mean_estimate": -0.00590314, + "mean_truth": -0.00054461, + "n_replicates": 3, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00277577, + "spread": 0.01770315, + "score": 0.01791945, + "mean_estimate": -0.01722366, + "mean_truth": -0.01999942, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.0058608, + "spread": 0.02123806, + "score": 0.02203189, + "mean_estimate": -0.02584341, + "mean_truth": -0.01998261, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00201022, + "spread": 0.00294831, + "score": 0.00356841, + "mean_estimate": -0.01637903, + "mean_truth": -0.01436881, + "n_replicates": 4, + "wall_time_s_sum": 0.3251883, + "wall_time_s_mean": 0.08129708 + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02493509, + "spread": 0.04100636, + "score": 0.0479925, + "mean_estimate": 0.00344385, + "mean_truth": -0.02149123, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00193289, + "spread": 0.01045667, + "score": 0.01063381, + "mean_estimate": -0.02144804, + "mean_truth": -0.01951516, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.01128722, + "spread": 0.02337352, + "score": 0.02595617, + "mean_estimate": 7.376e-05, + "mean_truth": -0.01121346, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.0175028, + "spread": 0.0288487, + "score": 0.03374308, + "mean_estimate": 0.01688458, + "mean_truth": -0.00061822, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00172356, + "spread": 0.01358396, + "score": 0.01369287, + "mean_estimate": -0.021723, + "mean_truth": -0.01999945, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00434002, + "spread": 0.01604773, + "score": 0.01662424, + "mean_estimate": -0.0243231, + "mean_truth": -0.01998307, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00067687, + "spread": 0.00325169, + "score": 0.00332139, + "mean_estimate": -0.01479789, + "mean_truth": -0.01412102, + "n_replicates": 4, + "wall_time_s_sum": 0.4099183, + "wall_time_s_mean": 0.10247957 + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00766804, + "spread": 0.0181085, + "score": 0.01966512, + "mean_estimate": -0.01362507, + "mean_truth": -0.02129311, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00284368, + "spread": 0.00807415, + "score": 0.00856028, + "mean_estimate": -0.01667436, + "mean_truth": -0.01951804, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00809728, + "spread": 0.00874476, + "score": 0.01191792, + "mean_estimate": -0.00312423, + "mean_truth": -0.0112215, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00319274, + "spread": 0.00144771, + "score": 0.00350563, + "mean_estimate": 0.00262817, + "mean_truth": -0.00056457, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00161083, + "spread": 0.00872783, + "score": 0.00887523, + "mean_estimate": -0.01838864, + "mean_truth": -0.01999947, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00603183, + "spread": 0.0123494, + "score": 0.01374375, + "mean_estimate": -0.02601535, + "mean_truth": -0.01998352, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00664203, + "spread": 0.01097593, + "score": 0.01282917, + "mean_estimate": 0.00970029, + "mean_truth": 0.01634232, + "n_replicates": 4, + "wall_time_s_sum": 0.1626668, + "wall_time_s_mean": 0.0406667 + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.01515681, + "spread": 0.04496994, + "score": 0.0474555, + "mean_estimate": 0.04869859, + "mean_truth": 0.03354178, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.01285815, + "spread": 0.02598986, + "score": 0.02899663, + "mean_estimate": 0.03243263, + "mean_truth": 0.01957448, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.0100285, + "spread": 0.02799206, + "score": 0.02973426, + "mean_estimate": 0.00342605, + "mean_truth": 0.01345456, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.01119145, + "spread": 0.01377021, + "score": 0.01774449, + "mean_estimate": -0.01070606, + "mean_truth": 0.00048538, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.01214128, + "spread": 0.01098239, + "score": 0.01637142, + "mean_estimate": 0.00785813, + "mean_truth": 0.01999941, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.01648952, + "spread": 0.04493542, + "score": 0.04786539, + "mean_estimate": 0.00349369, + "mean_truth": 0.01998321, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00193638, + "spread": 0.00172931, + "score": 0.00259617, + "mean_estimate": 0.01660111, + "mean_truth": 0.01466473, + "n_replicates": 4, + "wall_time_s_sum": 0.1730529, + "wall_time_s_mean": 0.04326322 + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04864638, + "spread": 0.07177508, + "score": 0.08670716, + "mean_estimate": 0.07081023, + "mean_truth": 0.02216385, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.01278619, + "spread": 0.01127457, + "score": 0.01704708, + "mean_estimate": 0.03230923, + "mean_truth": 0.01952304, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00801535, + "spread": 0.0301598, + "score": 0.03120671, + "mean_estimate": 0.01985789, + "mean_truth": 0.01184254, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00386821, + "spread": 0.00684602, + "score": 0.00786328, + "mean_estimate": -0.00334578, + "mean_truth": 0.00054461, + "n_replicates": 3, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00288452, + "spread": 0.01842306, + "score": 0.01864751, + "mean_estimate": 0.02288394, + "mean_truth": 0.01999942, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.0062596, + "spread": 0.02203298, + "score": 0.02290491, + "mean_estimate": 0.01372301, + "mean_truth": 0.01998261, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": 2.333e-05, + "spread": 0.00266549, + "score": 0.00266559, + "mean_estimate": 0.01439215, + "mean_truth": 0.01436881, + "n_replicates": 4, + "wall_time_s_sum": 0.2221385, + "wall_time_s_mean": 0.05553462 + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0251516, + "spread": 0.04267597, + "score": 0.04953626, + "mean_estimate": 0.04664284, + "mean_truth": 0.02149123, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00443687, + "spread": 0.01111106, + "score": 0.01196417, + "mean_estimate": 0.01507828, + "mean_truth": 0.01951516, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00907109, + "spread": 0.02459878, + "score": 0.02621802, + "mean_estimate": 0.02028455, + "mean_truth": 0.01121346, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.01990041, + "spread": 0.03001456, + "score": 0.0360125, + "mean_estimate": 0.02051863, + "mean_truth": 0.00061822, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00179768, + "spread": 0.01413637, + "score": 0.01425021, + "mean_estimate": 0.01820176, + "mean_truth": 0.01999945, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00466799, + "spread": 0.01668583, + "score": 0.01732649, + "mean_estimate": 0.01531508, + "mean_truth": 0.01998307, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00135156, + "spread": 0.00407716, + "score": 0.00429534, + "mean_estimate": 0.01547258, + "mean_truth": 0.01412102, + "n_replicates": 4, + "wall_time_s_sum": 0.3203721, + "wall_time_s_mean": 0.08009303 + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00718836, + "spread": 0.01817231, + "score": 0.0195424, + "mean_estimate": 0.02848147, + "mean_truth": 0.02129311, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -2.839e-05, + "spread": 0.0082439, + "score": 0.00824395, + "mean_estimate": 0.01948965, + "mean_truth": 0.01951804, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00492379, + "spread": 0.00985887, + "score": 0.01102003, + "mean_estimate": 0.0161453, + "mean_truth": 0.0112215, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": 0.00415177, + "spread": 0.00228869, + "score": 0.00474081, + "mean_estimate": 0.00471634, + "mean_truth": 0.00056457, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00167314, + "spread": 0.00908267, + "score": 0.00923549, + "mean_estimate": 0.02167261, + "mean_truth": 0.01999947, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "toggle_specialist", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00648709, + "spread": 0.01288324, + "score": 0.01442429, + "mean_estimate": 0.01349643, + "mean_truth": 0.01998352, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + } + ] +} diff --git a/benchmarking/baselines/study_toggle_methods_compare_baseline_win32.json b/benchmarking/baselines/study_toggle_methods_compare_baseline_win32.json new file mode 100644 index 00000000..9301c451 --- /dev/null +++ b/benchmarking/baselines/study_toggle_methods_compare_baseline_win32.json @@ -0,0 +1,1287 @@ +{ + "schema": "toggle_methods_compare_baseline_v3", + "recorded_utc": "2026-07-15T13:10:41Z", + "git_commit": "de85f84", + "platform": "win32", + "cpu_count": null, + "python_version": null, + "lightgbm_version": null, + "n_replicates": 4, + "seed": 0, + "campaign_weeks": [ + 1, + 2, + 4, + 8 + ], + "profiles": [ + "cp_0pct", + "cp_minus_2pct", + "cp_plus_2pct" + ], + "methods": [ + "power_model" + ], + "cells": [ + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00082113, + "spread": 0.003644, + "score": 0.00373537, + "mean_estimate": 0.00082113, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 51.0641546, + "wall_time_s_mean": 12.76603865 + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0294864, + "spread": 0.02838875, + "score": 0.04093127, + "mean_estimate": 0.0294864, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.0037195, + "spread": 0.00482864, + "score": 0.00609512, + "mean_estimate": -0.0037195, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00077085, + "spread": 0.00119919, + "score": 0.00142558, + "mean_estimate": -0.00077085, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00131161, + "spread": 0.0028903, + "score": 0.00317398, + "mean_estimate": -0.00065581, + "mean_truth": 0.0, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00313034, + "spread": 0.00389745, + "score": 0.00499892, + "mean_estimate": 0.00313034, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00322371, + "spread": 0.00679269, + "score": 0.00751884, + "mean_estimate": -0.00322371, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00083803, + "spread": 0.00160037, + "score": 0.00180651, + "mean_estimate": -0.00083803, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 49.3195071, + "wall_time_s_mean": 12.32987678 + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04431103, + "spread": 0.0624749, + "score": 0.0765936, + "mean_estimate": 0.04431103, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00099704, + "spread": 0.01631588, + "score": 0.01634631, + "mean_estimate": 0.00099704, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00561889, + "spread": 0.00785768, + "score": 0.00965997, + "mean_estimate": -0.00561889, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00145637, + "spread": 0.00152003, + "score": 0.00210512, + "mean_estimate": -0.00145637, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00076438, + "spread": 0.0130581, + "score": 0.01308045, + "mean_estimate": -0.00076438, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00227678, + "spread": 0.006137, + "score": 0.00654573, + "mean_estimate": -0.00227678, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00250329, + "spread": 0.00237335, + "score": 0.00344953, + "mean_estimate": -0.00250329, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 51.7736811, + "wall_time_s_mean": 12.94342027 + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02362597, + "spread": 0.03705684, + "score": 0.04394765, + "mean_estimate": 0.02362597, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00175174, + "spread": 0.01341645, + "score": 0.01353032, + "mean_estimate": 0.00175174, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00092146, + "spread": 0.01314207, + "score": 0.01317433, + "mean_estimate": -0.00092146, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.000527, + "spread": 0.00364999, + "score": 0.00368784, + "mean_estimate": -0.000527, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00142672, + "spread": 0.01193418, + "score": 0.01201916, + "mean_estimate": -0.00142672, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00840066, + "spread": 0.00839305, + "score": 0.01187495, + "mean_estimate": -0.00840066, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00117125, + "spread": 0.00198641, + "score": 0.002306, + "mean_estimate": -0.00117125, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": 61.4657606, + "wall_time_s_mean": 15.36644015 + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0112509, + "spread": 0.02031314, + "score": 0.02322082, + "mean_estimate": 0.0112509, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00093992, + "spread": 0.00662147, + "score": 0.00668784, + "mean_estimate": -0.00093992, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00120612, + "spread": 0.00338001, + "score": 0.00358876, + "mean_estimate": 0.00120612, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00430846, + "spread": 0.00388958, + "score": 0.00580445, + "mean_estimate": -0.00430846, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00280601, + "spread": 0.00316616, + "score": 0.00423063, + "mean_estimate": 0.00280601, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_0pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00466965, + "spread": 0.00274401, + "score": 0.0054162, + "mean_estimate": -0.00466965, + "mean_truth": 0.0, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00143225, + "spread": 0.00428401, + "score": 0.00451708, + "mean_estimate": -0.01491006, + "mean_truth": -0.01634232, + "n_replicates": 4, + "wall_time_s_sum": 49.3324591, + "wall_time_s_mean": 12.33311477 + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00028037, + "spread": 0.05037207, + "score": 0.05037285, + "mean_estimate": -0.03326141, + "mean_truth": -0.03354178, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00562711, + "spread": 0.0146591, + "score": 0.01570203, + "mean_estimate": -0.01394736, + "mean_truth": -0.01957448, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.0102508, + "spread": 0.00616335, + "score": 0.01196101, + "mean_estimate": -0.00320376, + "mean_truth": -0.01345456, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00285288, + "spread": 0.00323801, + "score": 0.00431551, + "mean_estimate": -0.00166913, + "mean_truth": -0.00048538, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00088596, + "spread": 0.02040366, + "score": 0.02042288, + "mean_estimate": -0.02088537, + "mean_truth": -0.01999941, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": 0.00734414, + "spread": 0.01354132, + "score": 0.01540467, + "mean_estimate": -0.01263907, + "mean_truth": -0.01998321, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00079856, + "spread": 0.00155716, + "score": 0.00174999, + "mean_estimate": -0.01546329, + "mean_truth": -0.01466473, + "n_replicates": 4, + "wall_time_s_sum": 50.8575067, + "wall_time_s_mean": 12.71437667 + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04098865, + "spread": 0.06442508, + "score": 0.07635876, + "mean_estimate": 0.0188248, + "mean_truth": -0.02216385, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00054201, + "spread": 0.01668047, + "score": 0.01668927, + "mean_estimate": -0.02006505, + "mean_truth": -0.01952304, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00117498, + "spread": 0.01149365, + "score": 0.01155355, + "mean_estimate": -0.01066756, + "mean_truth": -0.01184254, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00131714, + "spread": 0.00198179, + "score": 0.00237957, + "mean_estimate": -0.00186175, + "mean_truth": -0.00054461, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00119069, + "spread": 0.01185493, + "score": 0.01191457, + "mean_estimate": -0.02119012, + "mean_truth": -0.01999942, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00344004, + "spread": 0.00512394, + "score": 0.0061716, + "mean_estimate": -0.02342265, + "mean_truth": -0.01998261, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.0024563, + "spread": 0.00233048, + "score": 0.00338593, + "mean_estimate": -0.01682511, + "mean_truth": -0.01436881, + "n_replicates": 4, + "wall_time_s_sum": 58.439074, + "wall_time_s_mean": 14.6097685 + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02506806, + "spread": 0.03452059, + "score": 0.04266239, + "mean_estimate": 0.00357683, + "mean_truth": -0.02149123, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00321066, + "spread": 0.01252727, + "score": 0.01293217, + "mean_estimate": -0.0163045, + "mean_truth": -0.01951516, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00030456, + "spread": 0.01298888, + "score": 0.01299245, + "mean_estimate": -0.0109089, + "mean_truth": -0.01121346, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00087577, + "spread": 0.00387194, + "score": 0.00396975, + "mean_estimate": -0.00149398, + "mean_truth": -0.00061822, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00228055, + "spread": 0.0113767, + "score": 0.01160302, + "mean_estimate": -0.02227999, + "mean_truth": -0.01999945, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.0087042, + "spread": 0.00929069, + "score": 0.01273106, + "mean_estimate": -0.02868728, + "mean_truth": -0.01998307, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00114025, + "spread": 0.00195579, + "score": 0.00226391, + "mean_estimate": -0.01526127, + "mean_truth": -0.01412102, + "n_replicates": 4, + "wall_time_s_sum": 62.2375124, + "wall_time_s_mean": 15.5593781 + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.0092961, + "spread": 0.01842721, + "score": 0.02063927, + "mean_estimate": -0.011997, + "mean_truth": -0.02129311, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00019742, + "spread": 0.00687805, + "score": 0.00688088, + "mean_estimate": -0.01971547, + "mean_truth": -0.01951804, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00289289, + "spread": 0.00318311, + "score": 0.00430128, + "mean_estimate": -0.00832861, + "mean_truth": -0.0112215, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00455584, + "spread": 0.00440567, + "score": 0.00633764, + "mean_estimate": -0.00512041, + "mean_truth": -0.00056457, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00266982, + "spread": 0.0040223, + "score": 0.00482771, + "mean_estimate": -0.01732965, + "mean_truth": -0.01999947, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_minus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.0048403, + "spread": 0.00237248, + "score": 0.00539047, + "mean_estimate": -0.02482382, + "mean_truth": -0.01998352, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.00120204, + "spread": 0.00447419, + "score": 0.00463285, + "mean_estimate": 0.01754436, + "mean_truth": 0.01634232, + "n_replicates": 4, + "wall_time_s_sum": 47.8747112, + "wall_time_s_mean": 11.9686778 + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02712657, + "spread": 0.02948606, + "score": 0.04006593, + "mean_estimate": 0.06066835, + "mean_truth": 0.03354178, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.01378138, + "spread": 0.00610632, + "score": 0.0150736, + "mean_estimate": 0.0057931, + "mean_truth": 0.01957448, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.0116129, + "spread": 0.00463536, + "score": 0.01250384, + "mean_estimate": 0.00184166, + "mean_truth": 0.01345456, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00093943, + "spread": 0.00261338, + "score": 0.0027771, + "mean_estimate": -0.00022702, + "mean_truth": 0.00048538, + "n_replicates": 2, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.02506352, + "spread": 0.01830669, + "score": 0.03103731, + "mean_estimate": 0.04506292, + "mean_truth": 0.01999941, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 1, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.01189384, + "spread": 0.00954244, + "score": 0.01524866, + "mean_estimate": 0.00808937, + "mean_truth": 0.01998321, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00087189, + "spread": 0.00163846, + "score": 0.00185601, + "mean_estimate": 0.01379283, + "mean_truth": 0.01466473, + "n_replicates": 4, + "wall_time_s_sum": 60.572831, + "wall_time_s_mean": 15.14320775 + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.04618919, + "spread": 0.06332516, + "score": 0.0783806, + "mean_estimate": 0.06835304, + "mean_truth": 0.02216385, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00263676, + "spread": 0.01826356, + "score": 0.01845292, + "mean_estimate": 0.0221598, + "mean_truth": 0.01952304, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.01166903, + "spread": 0.00539377, + "score": 0.01285531, + "mean_estimate": 0.00017351, + "mean_truth": 0.01184254, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00212193, + "spread": 0.0017754, + "score": 0.0027667, + "mean_estimate": -0.00157732, + "mean_truth": 0.00054461, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00099845, + "spread": 0.01195955, + "score": 0.01200115, + "mean_estimate": 0.02099788, + "mean_truth": 0.01999942, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 2, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00181589, + "spread": 0.00924218, + "score": 0.00941889, + "mean_estimate": 0.01816673, + "mean_truth": 0.01998261, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00255028, + "spread": 0.00241623, + "score": 0.00351314, + "mean_estimate": 0.01181853, + "mean_truth": 0.01436881, + "n_replicates": 4, + "wall_time_s_sum": 61.7226545, + "wall_time_s_mean": 15.43066363 + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.02537651, + "spread": 0.03711936, + "score": 0.04496459, + "mean_estimate": 0.04686774, + "mean_truth": 0.02149123, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": 0.00031284, + "spread": 0.01399758, + "score": 0.01400107, + "mean_estimate": 0.01982799, + "mean_truth": 0.01951516, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": -0.00187071, + "spread": 0.01166222, + "score": 0.01181131, + "mean_estimate": 0.00934274, + "mean_truth": 0.01121346, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00050262, + "spread": 0.00409784, + "score": 0.00412855, + "mean_estimate": 0.0001156, + "mean_truth": 0.00061822, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": -0.00088927, + "spread": 0.01222771, + "score": 0.01226001, + "mean_estimate": 0.01911017, + "mean_truth": 0.01999945, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 4, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00828775, + "spread": 0.00857027, + "score": 0.01192209, + "mean_estimate": 0.01169533, + "mean_truth": 0.01998307, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "overall", + "condition_bin": "overall", + "bias": -0.00120224, + "spread": 0.00201706, + "score": 0.00234817, + "mean_estimate": 0.01291878, + "mean_truth": 0.01412102, + "n_replicates": 4, + "wall_time_s_sum": 65.5466537, + "wall_time_s_mean": 16.38666343 + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(-230.0, 230.0]", + "bias": 0.00827707, + "spread": 0.02037662, + "score": 0.02199356, + "mean_estimate": 0.02957018, + "mean_truth": 0.02129311, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1150.0, 1610.0]", + "bias": -0.00100285, + "spread": 0.00785163, + "score": 0.00791542, + "mean_estimate": 0.01851519, + "mean_truth": 0.01951804, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(1610.0, 2070.0]", + "bias": 0.00048196, + "spread": 0.00258823, + "score": 0.00263272, + "mean_estimate": 0.01170346, + "mean_truth": 0.0112215, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(2070.0, 2530.0]", + "bias": -0.00423112, + "spread": 0.00405175, + "score": 0.00585825, + "mean_estimate": -0.00366655, + "mean_truth": 0.00056457, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(230.0, 690.0]", + "bias": 0.00247973, + "spread": 0.00433681, + "score": 0.0049957, + "mean_estimate": 0.0224792, + "mean_truth": 0.01999947, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + }, + { + "method": "power_model", + "profile": "cp_plus_2pct", + "campaign_weeks": 8, + "condition": "power", + "condition_bin": "(690.0, 1150.0]", + "bias": -0.00447092, + "spread": 0.00278849, + "score": 0.00526923, + "mean_estimate": 0.0155126, + "mean_truth": 0.01998352, + "n_replicates": 4, + "wall_time_s_sum": NaN, + "wall_time_s_mean": NaN + } + ] +} diff --git a/benchmarking/baselines/study_toggle_specialist_uncertainty.py b/benchmarking/baselines/study_toggle_specialist_uncertainty.py new file mode 100644 index 00000000..29accf35 --- /dev/null +++ b/benchmarking/baselines/study_toggle_specialist_uncertainty.py @@ -0,0 +1,433 @@ +"""Measure how well ``toggle_specialist``'s reported 1-sigma matches its actual error. + +The P50 studies ask how close the estimate lands. This one asks whether the method was **right +about how close it would land** — the quality test of an uncertainty is coverage against ground +truth: ~68.3% of estimates should sit within 1 sigma of truth. Note that this scores sigma against +the *total* deviation from truth, bias included; a block bootstrap sees only sampling variance, so +where the method is biased the sigma will under-cover, and that is a finding rather than an +unfairness (see :mod:`benchmarking.harness.calibration`). + +Three design points, each with evidence behind it in F28-F30: + +**Replicates, not cells, are the evidence.** The three profiles reuse the same +``(turbine, treatment_start)`` draws, so their errors correlate 0.977-0.995; campaign lengths are +prefix-nested; long campaigns overlap each other. Coverage SE is therefore quoted on the independent +draw count, never on the row count — see :func:`~benchmarking.harness.coverage_standard_error`. + +**Block length is swept as a method variant.** Uncertainty never changes uplift and an estimate is +cheap, so one method per block length (:data:`BLOCK_HOURS_GRID`) gets the sweep from the existing +multi-method seam. There is no sigma-vs-L plateau to read here; coverage against truth decides. + +**Memory forces a streaming loop.** A replicate carries a ``synthetic_df`` and an ``original_df`` +(~0.5 GB), so ``score_study``'s materialised ensemble would be OOM-killed at this replicate count. +This drives a replicate-outer loop over :func:`~benchmarking.harness.iter_replicates`, reusing +:func:`~benchmarking.harness.score_one` so the truth alignment is shared rather than reimplemented. + +``cases.csv`` is the point of the run: every scored cell with its estimate, truth, sigma and record +counts, so a further uncertainty component can be fitted offline without re-running the sweep. + +Run from the repo root:: + + uv run python -m benchmarking.baselines.study_toggle_specialist_uncertainty + +Use ``--replicates 8 --profiles cp_0pct`` for a fast smoke run. +""" + +from __future__ import annotations + +import argparse +import logging +from pathlib import Path + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +from benchmarking.baselines.example_prepost_study import ( + DEFAULT_END_DT_EXCL, + DEFAULT_START_DT, + DEFAULT_TURBINE_SUBSET, + DEFAULT_WTG_NUMBERS, +) +from benchmarking.baselines.example_toggle_study import DEFAULT_TOGGLE_PERIOD +from benchmarking.baselines.study_toggle_methods_compare import TOGGLE_PROFILES +from benchmarking.baselines.toggle_specialist import DEFAULT_BLOCK_HOURS, ToggleSpecialistMethod +from benchmarking.harness import ( + TARGET_COVERAGE_1SIGMA, + StudyConfig, + campaign_windows, + coverage_standard_error, + iter_replicates, + score_one, + summarize_calibration, + truth_mask, +) +from benchmarking.synthetic import HOT_COLUMNS, HOT_RATED_POWER_KW +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +CAMPAIGN_WEEKS = [1, 2, 4, 8, 26, 52] +N_REPLICATES = 64 +SEED = 0 +# toggle_specialist drops pre-campaign rows (`restrict_to_campaign`), so a pre-campaign baseline buys +# it nothing and `min_pre_months` only costs start-range span. Widening the range to the whole dataset +# doubles the non-overlapping positions available to the long campaigns, which is where independent +# evidence is scarcest: a 52-week campaign is 364d, so a 730d range holds only ~2 of them. +TREATMENT_START_RANGE = (DEFAULT_START_DT, pd.Timestamp("2020-01-01", tz="UTC")) +MIN_PRE_MONTHS_TOGGLE = 0 +# Brackets the default on both sides, because coverage degrades in both directions (F28) and a grid +# that only reached upwards would hide half of that. The bottom end (1h ~ 1.5 cycles of a 40-minute +# toggle) over-covers; the top end starves the bootstrap of distinct blocks and biases sigma low. +# 96h is dropped: at 1 week it is ~2 blocks and its verdict (coverage 0.438) is already recorded. +BLOCK_HOURS_GRID = [1.0, 2.0, 3.0, 6.0, 12.0, 24.0, 48.0] + +_LENGTH_COL = "campaign_weeks" +_DEFAULT_OUTPUT_DIR = Path.home() / "temp" / "wind-up-benchmarking" / "toggle_specialist_uncertainty" +_PP = 100.0 # fraction -> percentage points +# Record-count buckets for the per-bin coverage-vs-count read, roughly by decade because the +# hypothesis is about order of magnitude ("sparse bins fail"), not about a particular count. +_COUNT_BIN_EDGES = (0, 30, 100, 300, 1000, 3000, 10000) + + +def uncertainty_study(n_replicates: int) -> StudyConfig: + """Return the study every run scores: a toggle grid from one week to a year.""" + return StudyConfig( + mode="toggle", + turbine_subset=DEFAULT_TURBINE_SUBSET, + treatment_start_range=TREATMENT_START_RANGE, + min_pre_months=MIN_PRE_MONTHS_TOGGLE, + campaign_weeks=CAMPAIGN_WEEKS, + toggle_period=DEFAULT_TOGGLE_PERIOD, + n_replicates=n_replicates, + seed=SEED, + ) + + +def independent_draws(campaign_weeks: int, *, n_replicates: int, n_turbines: int = len(DEFAULT_TURBINE_SUBSET)) -> int: + """Roughly how many *independent* cases a campaign length really has. + + Replicates are only independent while their windows do not overlap. A 52-week campaign drawn from + a ~4-year start range has ~4 non-overlapping positions, so 64 replicates carry ~4 x n_turbines + draws, not 64 — quoting coverage SE on 64 there would understate it ~2x. Capped at + ``n_replicates``, since drawing more positions than replicates buys nothing. + """ + lo, hi = TREATMENT_START_RANGE + positions = max((hi - lo) / pd.Timedelta(weeks=campaign_weeks), 1.0) + return int(min(n_replicates, round(positions * n_turbines))) + + +def build_methods(block_hours_grid: list[float], out_dir: Path | None = None) -> list[ToggleSpecialistMethod]: + """One ``toggle_specialist`` per block length, identical in every other respect. + + They therefore produce identical uplifts and differ only in sigma, which is what makes the + block-length sweep a method comparison the existing harness already knows how to run. + + ``out_dir`` is left ``None`` by default: the sweep would otherwise write thousands of per-run + diagnostic folders nothing reads, and its actual output is ``cases.csv``. + """ + return [ + ToggleSpecialistMethod( + columns=HOT_COLUMNS, + name=f"toggle_specialist_bl{block_hours:g}", + conditions=("power",), + rated_power_kw=HOT_RATED_POWER_KW, + block_hours=block_hours, + out_dir=out_dir, + ) + for block_hours in block_hours_grid + ] + + +def _select_profiles(requested: list[str] | None) -> dict[str, list]: + """Return the profiles to score: all when ``requested`` is ``None``, else the named subset.""" + if requested is None: + return TOGGLE_PROFILES + unknown = [name for name in requested if name not in TOGGLE_PROFILES] + if unknown: + msg = f"unknown profile(s) {unknown}; available: {sorted(TOGGLE_PROFILES)}" + raise ValueError(msg) + return {name: TOGGLE_PROFILES[name] for name in requested} + + +def run_sweep( + scada_df: pd.DataFrame, + *, + study: StudyConfig, + methods: list[ToggleSpecialistMethod], + profiles: dict[str, list], +) -> pd.DataFrame: + """Score every method over every profile, streaming replicates to bound memory. + + Mirrors :func:`~benchmarking.harness.score_study` but loops replicate-outer, so exactly one + replicate is alive at a time. Every method still sees the identical ``MethodInput`` for an + instance (it is built once per instance and shared), so the cross-method fairness that matters + for the block-length comparison is preserved. + """ + data_start, data_end = scada_df.index.min(), scada_df.index.max() + rows: list[dict[str, object]] = [] + for profile_name, profile in profiles.items(): + for replicate in iter_replicates(scada_df, profile=profile, study=study): + windows = campaign_windows( + replicate.treatment_start, + min_pre_months=study.min_pre_months, + campaign_months=study.campaign_months, + campaign_weeks=study.campaign_weeks, + data_start=data_start, + data_end=data_end, + ) + for window in windows: + mask = truth_mask(replicate, window) + truth = replicate.true_uplift(mask=mask).overall + for method in methods: + rows.extend( + score_one( + method, + replicate=replicate, + window=window, + truth=truth, + mask=mask, + profile_name=profile_name, + ) + ) + logger.info("scored %s replicate %d (%d rows so far)", profile_name, replicate.replicate_id, len(rows)) + frame = pd.DataFrame(rows) + return frame.assign(block_hours=frame["method"].map(_block_hours_of)) + + +def _block_hours_of(method_name: str) -> float: + """Recover a variant's block length from its name (``toggle_specialist_bl48`` -> 48.0).""" + return float(method_name.rsplit("_bl", 1)[1]) + + +def calibration_tables(cases: pd.DataFrame, *, n_replicates: int) -> dict[str, pd.DataFrame]: + """Reduce the scored cases to the calibration reads worth looking at. + + Headline and per-bin are split because they fail for different reasons. The by-length table also + carries its own ``n_independent`` / ``coverage_se``, since long campaigns overlap and are far + weaker evidence than their row count suggests (:func:`independent_draws`). + """ + headline = cases[cases["condition"] == "overall"] + per_bin = cases[cases["condition"] != "overall"] + by_length = summarize_calibration(headline, group_keys=["block_hours", _LENGTH_COL]) + # Per-length SE, because the independent-draw count collapses as campaigns lengthen and overlap: + # a flat SE would make the 52-week reads look far firmer than they are. + draws = by_length[_LENGTH_COL].map(lambda w: independent_draws(int(w), n_replicates=n_replicates)) + by_length = by_length.assign(n_independent=draws, coverage_se=draws.map(coverage_standard_error)) + return { + "headline_by_block": summarize_calibration(headline, group_keys=["block_hours"]), + "headline_by_block_and_length": by_length, + "headline_by_block_and_profile": summarize_calibration(headline, group_keys=["block_hours", "profile"]), + "per_bin_by_block": summarize_calibration(per_bin, group_keys=["block_hours"]), + "per_bin_by_block_and_bin": summarize_calibration(per_bin, group_keys=["block_hours", "condition_bin"]), + "per_bin_by_block_and_length": summarize_calibration(per_bin, group_keys=["block_hours", _LENGTH_COL]), + } + + +def plot_results(cases: pd.DataFrame, tables: dict[str, pd.DataFrame], out_dir: Path) -> None: + """Write the four plots that carry the findings.""" + out_dir.mkdir(parents=True, exist_ok=True) + focus = _focus_block_hours(cases) + _plot_coverage_by_length(tables["headline_by_block_and_length"], out_dir / "coverage_by_campaign_length.png") + _plot_sigma_plateau(cases, out_dir / "sigma_vs_block_length.png") + _plot_error_vs_sigma(cases, out_dir / "error_vs_sigma.png", block_hours=focus) + _plot_coverage_vs_count(cases, out_dir / "coverage_vs_record_count.png", block_hours=focus) + + +def _focus_block_hours(cases: pd.DataFrame) -> float: + """Return the block length the per-case plots show: the method default, else the nearest swept. + + ``--block-hours`` is a free grid, so the default need not be in it. Picking the nearest length + actually present keeps those plots meaningful for any grid, rather than silently emptying them. + """ + present = np.sort(cases["block_hours"].unique()) + if DEFAULT_BLOCK_HOURS in present: + return DEFAULT_BLOCK_HOURS + return float(present[np.argmin(np.abs(present - DEFAULT_BLOCK_HOURS))]) + + +def _plot_coverage_by_length(table: pd.DataFrame, path: Path) -> None: + """Headline coverage against campaign length, one line per block length.""" + fig, ax = plt.subplots(figsize=(9, 5)) + for block_hours, group in table.groupby("block_hours"): + ax.plot(group[_LENGTH_COL], group["coverage_1sigma"], marker="o", label=f"{block_hours:g}h") + ax.axhline(TARGET_COVERAGE_1SIGMA, color="k", linestyle="--", label="target 0.683") + ax.set_xlabel("campaign length [weeks]") + ax.set_ylabel("coverage at 1 sigma") + ax.set_title("Headline coverage vs campaign length (below the line = sigma too small)") + ax.set_ylim(0.0, 1.0) + ax.grid(visible=True, alpha=0.3) + ax.legend(title="block length") + _save(fig, path) + + +def _plot_sigma_plateau(cases: pd.DataFrame, path: Path) -> None: + """Mean headline sigma against block length, one line per campaign length. + + There is no plateau to read here (F28): the curve is flat to falling. The measured RMS error is + drawn alongside as the level sigma should reach, which makes the long-block collapse legible. + """ + headline = cases[cases["condition"] == "overall"] + fig, ax = plt.subplots(figsize=(9, 5)) + for length, group in headline.groupby(_LENGTH_COL): + by_block = group.groupby("block_hours") + line = ax.plot( + by_block["sigma"].mean().index, + by_block["sigma"].mean().to_numpy() * _PP, + marker="o", + label=f"{length}w sigma", + ) + rms = float(np.sqrt(np.mean(group["signed_error"].dropna().to_numpy() ** 2))) * _PP + ax.axhline(rms, color=line[0].get_color(), linestyle=":", linewidth=1.2) + ax.set_xscale("log") + ax.set_xlabel("block length [h] (log scale)") + ax.set_ylabel("mean sigma [pp]") + ax.set_title("Sigma vs block length; dotted = that campaign's actual RMS error (the level to reach)") + ax.grid(visible=True, alpha=0.3) + ax.legend() + _save(fig, path) + + +def _plot_error_vs_sigma(cases: pd.DataFrame, path: Path, *, block_hours: float) -> None: + """Absolute error against reported sigma for the headline, at one block length.""" + focus = cases[(cases["condition"] == "overall") & (cases["block_hours"] == block_hours)] + if focus.empty: + return + fig, ax = plt.subplots(figsize=(7, 7)) + for length, group in focus.groupby(_LENGTH_COL): + ax.scatter(group["sigma"] * _PP, group["signed_error"].abs() * _PP, s=14, alpha=0.6, label=f"{length}w") + lim = float(np.nanmax([focus["sigma"].max(), focus["signed_error"].abs().max()])) * _PP * 1.05 + ax.plot([0, lim], [0, lim], color="k", linestyle="--", linewidth=1, label="|error| = sigma") + ax.set_xlabel("reported sigma [pp]") + ax.set_ylabel("|signed error| [pp]") + ax.set_title(f"Headline |error| vs reported sigma ({block_hours:g}h blocks); ~68% should fall below the line") + ax.grid(visible=True, alpha=0.3) + ax.legend(title="campaign") + _save(fig, path) + + +def _plot_coverage_vs_count(cases: pd.DataFrame, path: Path, *, block_hours: float) -> None: + """Per-bin coverage against the bin's record count, at one block length. + + The hypothesis this plot exists to test: the bootstrap holds up where a bin is well populated + and fails where it is not, which would make record count the covariate of a further term. + """ + per_bin = cases[(cases["condition"] != "overall") & (cases["block_hours"] == block_hours)].copy() + usable = per_bin[np.isfinite(per_bin["sigma"]) & (per_bin["sigma"] > 0) & np.isfinite(per_bin["signed_error"])] + if usable.empty: + return + # Decade-ish edges, clipped to the counts actually present: a short or single-profile run does + # not reach the upper decades, and a fixed edge above the data makes pd.cut non-monotonic. + largest = int(usable["n_upgraded_records"].max()) + edges = [e for e in _COUNT_BIN_EDGES if e < largest] + [largest + 1] + usable = usable.assign( + count_bin=pd.cut(usable["n_upgraded_records"], bins=edges), + covered=(usable["signed_error"] / usable["sigma"]).abs() <= 1.0, + ) + grouped = usable.groupby("count_bin", observed=True) + coverage = grouped["covered"].mean() + counts = grouped.size() + + fig, (ax, ax_n) = plt.subplots(2, 1, sharex=True, figsize=(9, 7), height_ratios=[2, 1]) + x = np.arange(len(coverage)) + ax.plot(x, coverage.to_numpy(), marker="o", color="C1") + ax.axhline(TARGET_COVERAGE_1SIGMA, color="k", linestyle="--", label="target 0.683") + ax.set_ylabel("coverage at 1 sigma") + ax.set_ylim(0.0, 1.0) + ax.set_title(f"Per-bin coverage vs bin record count ({block_hours:g}h blocks)") + ax.grid(visible=True, alpha=0.3) + ax.legend() + ax_n.bar(x, counts.to_numpy(), color="C0", alpha=0.7) + ax_n.set_ylabel("cases") + ax_n.set_xlabel("upgraded records in bin") + ax_n.set_xticks(x) + ax_n.set_xticklabels([str(c) for c in coverage.index], rotation=20, ha="right", fontsize=8) + ax_n.grid(visible=True, alpha=0.3) + _save(fig, path) + + +def _save(fig: plt.Figure, path: Path) -> None: + """Write a figure to ``path`` (creating its folder) and close it.""" + path.parent.mkdir(parents=True, exist_ok=True) + fig.tight_layout() + fig.savefig(path, dpi=150) + plt.close(fig) + + +def _log_tables(tables: dict[str, pd.DataFrame], *, n_replicates: int) -> None: + """Log every calibration table, with the coverage target and per-length standard errors.""" + per_length = ", ".join( + f"{w}w ~{independent_draws(w, n_replicates=n_replicates)} draws " + f"(SE {coverage_standard_error(independent_draws(w, n_replicates=n_replicates)):.3f})" + for w in CAMPAIGN_WEEKS + ) + logger.info( + "Target coverage %.3f. Row counts below are cells, NOT independent samples: profiles share " + "campaign windows, campaign lengths are prefix-nested, and long campaigns overlap each other. " + "Independent draws per campaign length: %s.", + TARGET_COVERAGE_1SIGMA, + per_length, + ) + for name, table in tables.items(): + logger.info("%s:\n%s", name, table.round(4).to_string(index=False)) + + +def main() -> None: + """Score the block-length variants over the grid, then report and plot the calibration.""" + parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter) + parser.add_argument("--output-dir", type=Path, default=_DEFAULT_OUTPUT_DIR, help="where outputs are written") + parser.add_argument( + "--profiles", nargs="+", choices=sorted(TOGGLE_PROFILES), default=None, help="restrict to a profile subset" + ) + parser.add_argument("--replicates", type=int, default=N_REPLICATES, help="replicate ensemble size") + parser.add_argument( + "--block-hours", + nargs="+", + type=float, + default=BLOCK_HOURS_GRID, + help="block lengths to sweep, one method variant each", + ) + parser.add_argument( + "--save-run-dirs", + action="store_true", + help="also write each estimate's per-run diagnostic folder (thousands of them; off by default)", + ) + args = parser.parse_args() + + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s", force=True) + output_dir = args.output_dir.expanduser() + output_dir.mkdir(parents=True, exist_ok=True) + + scada_df, _ = load_hot_scada( + start_dt=DEFAULT_START_DT, + end_dt_excl=DEFAULT_END_DT_EXCL, + wtg_numbers=DEFAULT_WTG_NUMBERS, + wtg_names=DEFAULT_TURBINE_SUBSET, + ) + study = uncertainty_study(args.replicates) + methods = build_methods(args.block_hours, out_dir=output_dir / "runs" if args.save_run_dirs else None) + profiles = _select_profiles(args.profiles) + logger.info( + "Sweeping %d replicates x %d profiles x %d campaign lengths x %d block lengths = %d estimates", + args.replicates, + len(profiles), + len(CAMPAIGN_WEEKS), + len(methods), + args.replicates * len(profiles) * len(CAMPAIGN_WEEKS) * len(methods), + ) + + cases = run_sweep(scada_df, study=study, methods=methods, profiles=profiles) + cases_path = output_dir / "cases.csv" + cases.to_csv(cases_path, index=False) + logger.info("Wrote %d scored cells to %s", len(cases), cases_path) + + tables = calibration_tables(cases, n_replicates=args.replicates) + for name, table in tables.items(): + table.to_csv(output_dir / f"calibration_{name}.csv", index=False) + _log_tables(tables, n_replicates=args.replicates) + plot_results(cases, tables, output_dir / "plots") + logger.info("All done. Outputs under %s", output_dir) + + +if __name__ == "__main__": + main() diff --git a/benchmarking/baselines/time_features.py b/benchmarking/baselines/time_features.py new file mode 100644 index 00000000..5917e5f7 --- /dev/null +++ b/benchmarking/baselines/time_features.py @@ -0,0 +1,233 @@ +"""Explicitly-constructed time features for ML uplift methods. + +The counterfactual power-model methods normally drop the timestamp entirely before +modelling: a bare timestamp is not a weather variable and folding it in naively risks +leaking the treatment (design-note SS3 -- anything that could reflect the upgrade must not +enter the model except through the treatment-invariant reference features). But time itself +carries real, physically meaningful signal that is *not* weather: reference-turbine +instrumentation drifts slowly over a campaign, the wind resource and air density have a +seasonal cycle, and the diurnal cycle (via solar heating -> boundary-layer shear/turbulence, +and directly via light for some effects) modulates conditions across the day. This module +gives a model an explicit, named handle on each of those axes instead of leaving it to guess +from an opaque clock value: + +* :func:`days_since_campaign_start` -- a continuous linear clock (in days, negative before + the campaign starts) for slow drift such as reference-anemometer calibration decay. +* :func:`season_sin_cos` -- a smooth, cyclical encoding of time-of-year (sin/cos pair, so the + model sees December and January as adjacent rather than as opposite ends of a 0-364 ramp). +* :func:`solar_altitude_azimuth` -- the sun's position (altitude plus a cyclical azimuth + encoding), a proxy for diurnal heating and boundary-layer state that is far more + informative than clock hour alone because it also depends on latitude, longitude and + season. + +All functions are pure and vectorized over a :class:`pandas.DatetimeIndex` that must be +timezone-aware (the benchmarking harness works exclusively in UTC). + +The solar-position calculation implements the NOAA solar-position algorithm (the public NOAA +Solar Calculator spreadsheet equations, itself based on Jean Meeus's *Astronomical +Algorithms*). It deliberately omits the atmospheric refraction correction that the NOAA +spreadsheet applies near the horizon -- refraction only matters for altitudes within a couple +of degrees of the horizon, and skipping it keeps the implementation a direct, checkable +transcription of the core geometry. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd + +# Vocabulary of configurable time-feature groups; "season" and "solar" each expand to +# multiple columns (see season_sin_cos / solar_altitude_azimuth). +TIME_FEATURE_NAMES: tuple[str, ...] = ("days_since_campaign_start", "season", "solar") + +# Day-of-year of June 21st in a non-leap year (one later in leap years); the season anchor. +_JUNE21_DAY_OF_YEAR_NON_LEAP = 172 +# Mean length of a year in days (Gregorian calendar average, i.e. including leap years). +_DAYS_PER_YEAR = 365.25 +_SECONDS_PER_DAY = 86_400.0 +_MINUTES_PER_DAY = 1_440.0 + +# Julian date of the J2000.0 epoch (2000-01-01 12:00 UTC), the NOAA algorithm's time origin. +_J2000_JULIAN_DATE = 2_451_545.0 +# Julian days per Julian century, used to convert Julian date into Julian centuries since J2000. +_JULIAN_DAYS_PER_CENTURY = 36_525.0 + +# Degrees of hour angle per minute of time (360 degrees / 1440 minutes per day). +_DEGREES_PER_MINUTE = 0.25 +_MINUTES_PER_DEGREE_LONGITUDE = 4.0 +_DEGREES_PER_CIRCLE = 360.0 + + +def _require_tz_aware(index: pd.DatetimeIndex) -> None: + """Raise ``ValueError`` unless ``index`` is timezone-aware. + + :param index: the index to check. + """ + if index.tz is None: + msg = "index must be timezone-aware (expected tz-aware UTC); got a tz-naive DatetimeIndex." + raise ValueError(msg) + + +def days_since_campaign_start(index: pd.DatetimeIndex, *, campaign_start: pd.Timestamp) -> pd.Series: + """Continuous days elapsed since ``campaign_start``, negative before it. + + A linear clock feature for slow, monotonic drift (e.g. reference-anemometer calibration + decay) that a purely cyclical feature such as :func:`season_sin_cos` cannot represent. + + :param index: timezone-aware timestamps to featurize. + :param campaign_start: the reference instant; must be comparable to ``index`` (i.e. also + timezone-aware). + :return: a :class:`pandas.Series` named ``"days_since_campaign_start"``, indexed by + ``index``, holding ``(index - campaign_start) / 1 day`` as a float. + """ + _require_tz_aware(index) + if campaign_start.tzinfo is None: + msg = "campaign_start must be timezone-aware (expected tz-aware UTC); got a tz-naive Timestamp." + raise ValueError(msg) + days = (index - campaign_start) / pd.Timedelta(days=1) + return pd.Series(days, index=index, name="days_since_campaign_start") + + +def season_sin_cos(index: pd.DatetimeIndex) -> pd.DataFrame: + """Time-of-year as a sin/cos pair anchored on the June 21st solstice. + + The angle is ``2*pi * (fractional day-of-year offset from June 21) / 365.25``, so the + encoding is smooth and cyclical (no discontinuity at year end) and ``season_cos`` peaks + at +1 around the June solstice and -1 around the December solstice, giving a tree-based + model a direct handle on the seasonal cycle without a sharp day-365-to-day-1 jump. + + :param index: timezone-aware timestamps to featurize. + :return: a :class:`pandas.DataFrame` with columns ``"season_sin"`` and ``"season_cos"``, + indexed by ``index``. + """ + _require_tz_aware(index) + index_utc = index.tz_convert("UTC") # wall-clock fields below must read UTC, whatever tz came in + fractional_day_of_year = ( + index_utc.dayofyear.to_numpy(dtype=float) + + ( + index_utc.hour.to_numpy(dtype=float) * 3600.0 + + index_utc.minute.to_numpy(dtype=float) * 60.0 + + index_utc.second.to_numpy(dtype=float) + + index_utc.microsecond.to_numpy(dtype=float) / 1.0e6 + ) + / _SECONDS_PER_DAY + ) + # June 21 is day 172 in a non-leap year but 173 in a leap year (the extra Feb 29 shifts it). + june21_day_of_year = _JUNE21_DAY_OF_YEAR_NON_LEAP + index_utc.is_leap_year.astype(float) + offset_from_june21 = fractional_day_of_year - june21_day_of_year + angle = 2.0 * np.pi * offset_from_june21 / _DAYS_PER_YEAR + return pd.DataFrame({"season_sin": np.sin(angle), "season_cos": np.cos(angle)}, index=index) + + +def solar_altitude_azimuth(index: pd.DatetimeIndex, *, latitude: float, longitude: float) -> pd.DataFrame: + """Solar altitude and azimuth via the NOAA solar-position algorithm. + + A vectorized (numpy-only, no per-row loops) transcription of the NOAA Solar Calculator + spreadsheet equations: julian day/century from the UTC timestamps, the sun's geometric + mean longitude and anomaly, the equation-of-center correction to true longitude, apparent + longitude, mean and corrected obliquity of the ecliptic, solar declination, the equation + of time, true solar time (from longitude and equation of time), hour angle, and finally + solar zenith (-> altitude) and azimuth. Atmospheric refraction near the horizon is + **not** applied, so altitude is the true geometric altitude rather than the + apparent/refracted one; this keeps the implementation a direct, checkable transcription + of the core geometry (refraction only matters within a couple of degrees of the horizon). + + :param index: timezone-aware timestamps to featurize (converted to UTC internally). + :param latitude: observer latitude in degrees, positive north. + :param longitude: observer longitude in degrees, positive east. + :return: a :class:`pandas.DataFrame` with columns ``"solar_altitude"`` (degrees, negative + below the horizon), ``"solar_azimuth_sin"`` and ``"solar_azimuth_cos"`` (sine and + cosine of the azimuth in radians, clockwise from north -- encoded cyclically because + azimuth wraps at 360 degrees and the downstream model is tree-based), indexed by + ``index``. + """ + _require_tz_aware(index) + index_utc = index.tz_convert("UTC") + + julian_date = index_utc.to_julian_date().to_numpy(dtype=float) + julian_century = (julian_date - _J2000_JULIAN_DATE) / _JULIAN_DAYS_PER_CENTURY + + geom_mean_long_sun = np.mod(280.46646 + julian_century * (36000.76983 + julian_century * 0.0003032), 360.0) + geom_mean_anom_sun = 357.52911 + julian_century * (35999.05029 - 0.0001537 * julian_century) + eccent_earth_orbit = 0.016708634 - julian_century * (0.000042037 + 0.0000001267 * julian_century) + + mean_anom_rad = np.radians(geom_mean_anom_sun) + sun_eq_of_ctr = ( + np.sin(mean_anom_rad) * (1.914602 - julian_century * (0.004817 + 0.000014 * julian_century)) + + np.sin(2.0 * mean_anom_rad) * (0.019993 - 0.000101 * julian_century) + + np.sin(3.0 * mean_anom_rad) * 0.000289 + ) + + sun_true_long = geom_mean_long_sun + sun_eq_of_ctr + sun_app_long = sun_true_long - 0.00569 - 0.00478 * np.sin(np.radians(125.04 - 1934.136 * julian_century)) + + mean_obliq_ecliptic = ( + 23.0 + + (26.0 + (21.448 - julian_century * (46.815 + julian_century * (0.00059 - julian_century * 0.001813))) / 60.0) + / 60.0 + ) + obliq_corr = mean_obliq_ecliptic + 0.00256 * np.cos(np.radians(125.04 - 1934.136 * julian_century)) + + sun_declin = np.degrees(np.arcsin(np.sin(np.radians(obliq_corr)) * np.sin(np.radians(sun_app_long)))) + + var_y = np.tan(np.radians(obliq_corr / 2.0)) ** 2 + geom_mean_long_sun_rad = np.radians(geom_mean_long_sun) + equation_of_time = 4.0 * np.degrees( + var_y * np.sin(2.0 * geom_mean_long_sun_rad) + - 2.0 * eccent_earth_orbit * np.sin(mean_anom_rad) + + 4.0 * eccent_earth_orbit * var_y * np.sin(mean_anom_rad) * np.cos(2.0 * geom_mean_long_sun_rad) + - 0.5 * var_y * var_y * np.sin(4.0 * geom_mean_long_sun_rad) + - 1.25 * eccent_earth_orbit * eccent_earth_orbit * np.sin(2.0 * mean_anom_rad) + ) + + minutes_since_midnight = ( + index_utc.hour.to_numpy(dtype=float) * 60.0 + + index_utc.minute.to_numpy(dtype=float) + + index_utc.second.to_numpy(dtype=float) / 60.0 + + index_utc.microsecond.to_numpy(dtype=float) / 60.0e6 + ) + true_solar_time = np.mod( + minutes_since_midnight + equation_of_time + _MINUTES_PER_DEGREE_LONGITUDE * longitude, _MINUTES_PER_DAY + ) + # true_solar_time is wrapped into [0, 1440) above, so the NOAA spreadsheet's negative branch + # cannot occur and the hour angle reduces to the single expression over [-180, 180). + hour_angle = true_solar_time * _DEGREES_PER_MINUTE - 180.0 + + lat_rad = np.radians(latitude) + declin_rad = np.radians(sun_declin) + hour_angle_rad = np.radians(hour_angle) + + cos_zenith = np.clip( + np.sin(lat_rad) * np.sin(declin_rad) + np.cos(lat_rad) * np.cos(declin_rad) * np.cos(hour_angle_rad), + -1.0, + 1.0, + ) + zenith = np.degrees(np.arccos(cos_zenith)) + altitude = 90.0 - zenith + + zenith_rad = np.radians(zenith) + with np.errstate(divide="ignore", invalid="ignore"): + azimuth_arg = np.clip( + (np.sin(lat_rad) * np.cos(zenith_rad) - np.sin(declin_rad)) / (np.cos(lat_rad) * np.sin(zenith_rad)), + -1.0, + 1.0, + ) + azimuth_base = np.degrees(np.arccos(azimuth_arg)) + azimuth = np.where( + hour_angle > 0.0, + np.mod(azimuth_base + 180.0, _DEGREES_PER_CIRCLE), + np.mod(540.0 - azimuth_base, _DEGREES_PER_CIRCLE), + ) + # Directly overhead (or a pole latitude), azimuth is undefined; the division above yields + # nan there, which np.mod propagates -- fall back to due-north (0 degrees) by convention. + azimuth = np.where(np.isnan(azimuth), 0.0, azimuth) + + azimuth_rad = np.radians(azimuth) + return pd.DataFrame( + { + "solar_altitude": altitude, + "solar_azimuth_sin": np.sin(azimuth_rad), + "solar_azimuth_cos": np.cos(azimuth_rad), + }, + index=index, + ) diff --git a/benchmarking/baselines/toggle_specialist.py b/benchmarking/baselines/toggle_specialist.py new file mode 100644 index 00000000..3cd678f7 --- /dev/null +++ b/benchmarking/baselines/toggle_specialist.py @@ -0,0 +1,872 @@ +"""A toggle-only energy-ratio uplift method behind the harness ``Method`` seam. + +``ToggleSpecialistMethod`` is a specialist for **toggle campaigns**: campaigns whose on/off +comparison is drawn entirely from the interleaved campaign blocks. It therefore accepts only +toggle inputs and raises on a prepost changeover. + +**Used timestamps** require every turbine (test and references) to be available (an availability +counter at a full period) and have finite power — a down turbine on either side of the ratio would +otherwise bias it. The availability column is therefore **required**. Only the active-power column +enters the ``rho`` *computation*; the availability column is used solely for row selection (cause, +not effect), so the estimate still +never conditions on the test turbine's post-treatment wind speed (design-note §3). It speaks the +data source's own column names and has no wind_up dependency. + +Every uplift — the headline and each power bin — comes with a non-optional 1-sigma uncertainty from +a circular block bootstrap (:mod:`benchmarking.baselines.block_bootstrap`). It is computed after the +uplift, from the uplift's own frozen row selection and bin assignment, and only when the uplift is +finite, so it cannot change any uplift result. The bootstrap sees sampling variability only, so +sigma under-covers where the method is biased (F29). + +Each run writes a per-run folder ``toggle_specialist___/`` +(v0-style naming) under ``out_dir`` (a temp dir by default), holding a per-segment data-stats CSV, +a headline results CSV, and -- when ``save_plots`` -- three diagnostic plots (a test-vs-reference +scatter, a per-segment daily-ratio timeseries, and a per-segment used-data-coverage timeseries). +The rich stats let a human confirm the right data was received and interpreted: the headline uplift +is re-derivable from the stats CSV as ``rho = used_test_mwh / used_ref_total_mwh`` per segment. +""" + +from __future__ import annotations + +import tempfile +from dataclasses import dataclass, replace +from pathlib import Path +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +from matplotlib.ticker import PercentFormatter + +from benchmarking.baselines.block_bootstrap import BootstrapResult, bootstrap_ratio_uplift +from benchmarking.baselines.filtering import NormalOperationFilter +from benchmarking.diagnostics import DiagnosticContext, stages, write_common_diagnostics, write_run_config +from benchmarking.harness.conditions import condition_bins, energy_ratio_by_bin, validate_conditions +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.harness.toggle import ToggleRowSets, is_toggle, resolve_toggle, toggle_upgrade_start +from benchmarking.synthetic import ToggleSchedule + +if TYPE_CHECKING: + import numpy.typing as npt + + from benchmarking.synthetic import ColumnSchema + +_SEGMENTS = ("all", "baseline", "upgraded") +_MIN_POINTS_FOR_TIMEBASE = 2 +# The cell name (and the ``(condition, condition_bin)`` key) of the headline uplift. +_OVERALL = "overall" + +# ``MethodOutput.labeled_rows`` segment labels. "excluded" = claimed by neither side (e.g. +# pre-campaign), which is distinct from a row in a segment that failed the filters (``used`` False). +_BASELINE = "baseline" +_UPGRADED = "upgraded" +_EXCLUDED = "excluded" +# Circular-block length for the uncertainty bootstrap, in hours. Must hold several on/off toggle +# cycles, so raise it for a campaign with a slow toggle period. +DEFAULT_BLOCK_HOURS = 6.0 +# ``power`` is the only axis this method can offer: it is derived from the references, so the +# treatment cannot move a row between bins. Binning by the test turbine's ws/TI would condition on +# post-treatment signals, which this method exists not to do (see the module docstring). +_SUPPORTED_CONDITIONS: tuple[str, ...] = ("power",) + + +def _infer_timebase(index: pd.DatetimeIndex) -> pd.Timedelta: + """Infer the analysis timebase as the median spacing of the sorted unique timestamps.""" + unique = pd.DatetimeIndex(pd.unique(index)).sort_values() + if len(unique) < _MIN_POINTS_FOR_TIMEBASE: + return pd.Timedelta(minutes=10) + return pd.Timedelta(np.median(np.diff(unique.to_numpy()))) + + +def _wide_column(scada_df: pd.DataFrame, *, turbine_col: str, value_col: str) -> pd.DataFrame: + """Pivot long SCADA to a timestamp x turbine table of ``value_col`` (NaN where missing).""" + tmp = scada_df[[turbine_col, value_col]].copy() + tmp["_ts"] = scada_df.index + return tmp.pivot_table( + index="_ts", + columns=turbine_col, + values=value_col, + aggfunc="first", + ) + + +def restrict_to_campaign(mi: MethodInput) -> MethodInput: + """Drop pre-campaign rows so the on/off comparison shares a distribution. + + The harness window can also carry a pre-campaign baseline, whose distribution differs from the + campaign and reintroduces the covariate shift toggling exists to avoid. Restrict the input to + records at/after the toggle start, leaving only the interleaved on/off blocks. A no-op when the + schedule has no explicit start (e.g. an already-campaign-only ``toggle_df``). + """ + timing = mi.upgrade_timing + if not (isinstance(timing, ToggleSchedule) and timing.start is not None): + return mi + return replace(mi, scada_df=mi.scada_df.loc[mi.scada_df.index >= timing.start]) + + +@dataclass +class ToggleSpecialistMethod: + """Pluggable toggle-only energy-ratio baseline. + + Accepts only toggle campaigns; ``estimate`` raises on a prepost changeover. Always fits on the + interleaved campaign on/off blocks, so on and off share a wind distribution. + + :param columns: **required** source-native column schema. Reads the ``active_power`` role (the + only signal in the ``rho`` computation) and the ``availability`` role (the required downtime + filter, applied to the test turbine and every reference); other roles feed diagnostics only. + :param name: method name shown in the leaderboard + :param out_dir: where per-run folders are written; a temp dir when ``None`` + :param save_plots: also write the diagnostic plots under ``/plots`` + :param timebase: analysis timebase; inferred from the data when ``None`` + :param conditions: condition axes to report a per-bin uplift over. Only ``"power"`` is supported + (see :meth:`_conditional_frame`); defaults to reporting none. + :param rated_power_kw: the test turbine's rated power, **required** when ``"power"`` is in + ``conditions``, since the power bin edges scale with the rating. + :param block_hours: circular-block length for the uncertainty bootstrap. Must hold several on/off + toggle cycles and stay a small fraction of the campaign; **raise it for a slow toggle + period**, for which :data:`DEFAULT_BLOCK_HOURS` may span only a cycle or two. + :param n_resamples: bootstrap resamples; block sums are precomputed, so this can be generous. + :param bootstrap_seed: RNG seed for the bootstrap, so a reported sigma is reproducible. + """ + + columns: ColumnSchema + name: str = "toggle_specialist" + out_dir: Path | None = None + save_plots: bool = False + timebase: pd.Timedelta | None = None + conditions: tuple[str, ...] = () + rated_power_kw: float | None = None + block_hours: float = DEFAULT_BLOCK_HOURS + n_resamples: int = 1000 + bootstrap_seed: int = 0 + + def __post_init__(self) -> None: + """Validate ``columns`` names every role this method reads, and the requested ``conditions``.""" + self.columns.require_roles(("active_power", "availability")) + validate_conditions(self.conditions, supported=_SUPPORTED_CONDITIONS, method_name=self.name) + if "power" in self.conditions and self.rated_power_kw is None: + msg = ( + f"{self.name}: rated_power_kw is required when 'power' is in conditions — the power bin " + f"edges are fractions of the turbine's rating." + ) + raise ValueError(msg) + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Estimate the test turbine's P50 uplift for one toggle campaign and write diagnostics.""" + if not is_toggle(mi.upgrade_timing): + msg = ( + f"ToggleSpecialistMethod only supports toggle campaigns, but upgrade_timing is a " + f"{type(mi.upgrade_timing).__name__} (a prepost changeover). Pass a toggle schedule " + f"or toggle_df; this method has no prepost baseline to compare against." + ) + raise ValueError(msg) + + mi = restrict_to_campaign(mi) + wide = _wide_column(mi.scada_df, turbine_col=mi.turbine_col, value_col=self.columns.active_power) + test = mi.test_wtg + refs = [c for c in wide.columns if c != test] + if not refs: + msg = ( + f"no reference turbines available for test_wtg {test!r}: scada_df contains only " + f"{list(wide.columns)}. The toggle specialist method needs at least one reference turbine." + ) + raise ValueError(msg) + + if self.columns.availability not in mi.scada_df.columns: + msg = ( + f"the availability column {self.columns.availability!r} (columns.availability) is not in " + f"scada_df; the downtime filter is required for the toggle specialist method and cannot be skipped." + ) + raise ValueError(msg) + + timebase = self.timebase if self.timebase is not None else _infer_timebase(mi.scada_df.index) + rows = resolve_toggle(mi.upgrade_timing, wide.index) + baseline = rows.campaign_baseline + test_pw = wide[test].to_numpy(dtype=float) + ref_total = wide[refs].sum(axis=1).to_numpy(dtype=float) + used = self._used_mask(mi, wide=wide, test=test, refs=refs, timebase=timebase).to_numpy() + + rho_base = _rho(test_pw, ref_total, used & baseline) + rho_up = _rho(test_pw, ref_total, used & rows.upgraded) + recoverable = np.isfinite(rho_base) and rho_base != 0 and np.isfinite(rho_up) + uplift = rho_up / rho_base - 1.0 if recoverable else np.nan + rho_label = _rho_label(rho_base, rho_up) + + per_bin = ( + self._conditional_frame( + test_pw=test_pw, + ref_total=ref_total, + rho_label=rho_label, + baseline=used & baseline, + upgraded=used & rows.upgraded, + ) + if "power" in self.conditions + else None + ) + + # Uncertainty runs strictly after the uplift, off the same frozen row selection and bin + # assignment, and only when there is a finite uplift to qualify. + membership = self._cell_membership(rho_label=rho_label, ref_total=ref_total, used=used) + boot = ( + self._bootstrap( + index=wide.index, + test_pw=test_pw, + ref_total=ref_total, + used=used, + upgraded=rows.upgraded, + baseline=baseline, + membership=membership, + timebase=timebase, + ) + if np.isfinite(uplift) + else None + ) + if per_bin is not None: + per_bin["sigma_uplift"] = [_cell_sigma(boot, str(b)) for b in per_bin["condition_bin"]] + diagnostics = _uncertainty_diagnostics( + boot, + membership=membership, + upgraded=used & rows.upgraded, + baseline=used & baseline, + used=used, + ) + + stats = _segment_stats( + mi, + wide=wide, + used=used, + toggle_rows=rows, + refs=refs, + timebase=timebase, + active_power_col=self.columns.active_power, + ) + sigma_overall = _cell_sigma(boot, _OVERALL) + self._write_outputs( + mi, + wide=wide, + stats=stats, + used=used, + rho_base=rho_base, + rho_up=rho_up, + uplift=uplift, + sigma_overall=sigma_overall, + n_refs=len(refs), + timebase=timebase, + per_bin=per_bin, + diagnostics=diagnostics, + ) + return MethodOutput( + p50_overall=float(uplift), + p50_by_condition=per_bin, + sigma_overall=sigma_overall, + uncertainty_diagnostics=diagnostics, + labeled_rows=self._labeled_rows( + mi, wide=wide, test=test, used=used, rows=rows, rho_label=rho_label, ref_total=ref_total + ), + ) + + def _labeled_rows( + self, + mi: MethodInput, + *, + wide: pd.DataFrame, + test: str, + used: npt.NDArray[np.bool_], + rows: ToggleRowSets, + rho_label: float, + ref_total: npt.NDArray[np.float64], + ) -> pd.DataFrame: + """Return the test turbine's own records, tagged with the labels this estimate was built from. + + Labels are reindexed from the arrays the uplift and bootstrap used, not recomputed, so an + aggregation of this frame lands on the same rows and bins the estimate did. + """ + labeled = mi.scada_df[mi.scada_df[mi.turbine_col] == test].copy() + + def _on_test_rows(values: npt.NDArray[np.generic]) -> npt.NDArray[np.generic]: + return pd.Series(values, index=wide.index).reindex(labeled.index).to_numpy() + + labeled["used"] = _on_test_rows(used) + labeled["segment"] = _on_test_rows( + np.where(rows.upgraded, _UPGRADED, np.where(rows.campaign_baseline, _BASELINE, _EXCLUDED)) + ) + + # The bin label is the same reference-derived baseline power the uplift binned on, so a row + # cannot sit in one bin here and another there. Outside the outer edges pd.cut gives NaN, + # which is carried through as "this row belongs to no bin" rather than clipped to an edge. + if "power" in self.conditions and np.isfinite(rho_label): + assert self.rated_power_kw is not None # noqa: S101 - guaranteed by __post_init__ + bins = condition_bins("power", rated_power_kw=self.rated_power_kw) + labeled["power_bin"] = _on_test_rows(np.asarray(pd.cut(rho_label * ref_total, bins=bins))) + return labeled + + def _cell_membership( + self, + *, + rho_label: float, + ref_total: npt.NDArray[np.float64], + used: npt.NDArray[np.bool_], + ) -> dict[str, npt.NDArray[np.bool_]]: + """Which **used** records belong to each bootstrap cell: the headline, plus each power bin. + + Reuses :meth:`_conditional_frame`'s own label and edges, so a record's cell is fixed by the + uplift computation and cannot move under resampling. + """ + used_idx = np.flatnonzero(used) + membership: dict[str, npt.NDArray[np.bool_]] = {_OVERALL: np.ones(len(used_idx), dtype=bool)} + if "power" not in self.conditions or not np.isfinite(rho_label): + return membership + assert self.rated_power_kw is not None # noqa: S101 - guaranteed by __post_init__ + bins = condition_bins("power", rated_power_kw=self.rated_power_kw) + assigned = pd.cut(rho_label * ref_total[used_idx], bins=bins) + for category in assigned.categories: + membership[str(category)] = np.asarray(assigned == category) + return membership + + def _bootstrap( + self, + *, + index: pd.DatetimeIndex, + test_pw: npt.NDArray[np.float64], + ref_total: npt.NDArray[np.float64], + used: npt.NDArray[np.bool_], + upgraded: npt.NDArray[np.bool_], + baseline: npt.NDArray[np.bool_], + membership: dict[str, npt.NDArray[np.bool_]], + timebase: pd.Timedelta, + ) -> BootstrapResult: + """Run the circular block bootstrap over the used records of the campaign. + + The campaign span is taken from the on/off rows rather than from ``index``, so blocks tile + the campaign itself even when the caller's window carries pre-campaign rows the estimate + never used. + """ + used_idx = np.flatnonzero(used) + campaign = upgraded | baseline + return bootstrap_ratio_uplift( + times=index[used_idx], + test_power=test_pw[used_idx], + ref_total=ref_total[used_idx], + upgraded=upgraded[used_idx], + baseline=baseline[used_idx], + cell_membership=membership, + campaign_start=index[campaign].min(), + campaign_end=index[campaign].max(), + timebase=timebase, + block_hours=self.block_hours, + n_resamples=self.n_resamples, + seed=self.bootstrap_seed, + ) + + def _conditional_frame( + self, + *, + test_pw: npt.NDArray[np.float64], + ref_total: npt.NDArray[np.float64], + rho_label: float, + baseline: npt.NDArray[np.bool_], + upgraded: npt.NDArray[np.bool_], + ) -> pd.DataFrame: + """Per-power-bin uplift: ``rho_up(b) / rho_base(b) - 1``, on bins of the mean operating point. + + Two decisions carry this, and both are needed: + + **The bin label is** ``rho_label * ref_total`` (see :func:`_rho_label`): reference-derived and + state-neutral, so neither the upgrade nor which state is called baseline can move a row + between bins, and it is on the test turbine's own kW scale. + + **The denominator is the per-bin** ``rho_base(b)``, not the global one: the test-to-reference + ratio varies with power, and a global denominator would read that structure as uplift. The + price is that the per-bin numbers no longer aggregate exactly to ``p50_overall``, which is + deliberate and un-relevelled; ``sum_actual`` / ``sum_counterfactual`` expose the gap. + + Sparse bins report NaN with ``n_records = 0`` rather than being imputed. + """ + assert self.rated_power_kw is not None # noqa: S101 - guaranteed by __post_init__ + bins = condition_bins("power", rated_power_kw=self.rated_power_kw) + label = rho_label * ref_total + counterfactual = _per_bin_counterfactual( + label=label, test_pw=test_pw, ref_total=ref_total, baseline=baseline, bins=bins + ) + frame = energy_ratio_by_bin(label[upgraded], test_pw[upgraded], counterfactual[upgraded], bins=bins) + frame.insert(0, "condition", "power") + return frame + + def _used_mask( + self, mi: MethodInput, *, wide: pd.DataFrame, test: str, refs: list[str], timebase: pd.Timedelta + ) -> pd.Series: + """Complete-case timestamps that also pass downtime filtering on the test turbine and every reference. + + Returns a bool Series on ``wide.index``. Every turbine (test and references) must be + available (counter >= a full period) and have finite power — a down turbine on either side + of the ratio is therefore excluded. The test turbine additionally goes through the shared + :class:`NormalOperationFilter` (the same downtime + finite-power logic the R-learner uses; + the stuck filter is left off here as the ratio sums raw power rather than fitting a model). + """ + turbines = [test, *refs] + complete = wide[turbines].notna().all(axis=1) + + full = timebase.total_seconds() + avail = _wide_column(mi.scada_df, turbine_col=mi.turbine_col, value_col=self.columns.availability).reindex( + index=wide.index, columns=turbines + ) + all_available = (avail >= full).all(axis=1) + + test_rows = mi.scada_df[mi.scada_df[mi.turbine_col] == test] + test_keep = ( + NormalOperationFilter( + active_power_col=self.columns.active_power, + availability_col=self.columns.availability, + apply_stuck_filter=False, + ) + .keep_mask(test_rows, timebase=timebase) + .reindex(wide.index, fill_value=False) + ) + return complete & all_available & test_keep & ~self._test_excluded(mi, test=test, index=wide.index) + + def _test_excluded(self, mi: MethodInput, *, test: str, index: pd.DatetimeIndex) -> pd.Series: + """Boolean mask on *index*: the test turbine's caller-flagged rows to drop (empty when unused). + + Reads the ``columns.exclude_row`` column of the test turbine's rows; references are never + excluded here (their special modes still carry information for the ratio). Absent column or + unset role -> nothing excluded. Reindex fills missing timestamps with ``False`` so an expanded + index never becomes an exclusion. NaN raises rather than coercing: ``astype(bool)`` reads a + missing flag as ``True`` and drops the row, the opposite of the safe default. + """ + col = self.columns.exclude_row + if not col or col not in mi.scada_df.columns: + return pd.Series(data=False, index=index, dtype=bool) + test_rows = mi.scada_df[mi.scada_df[mi.turbine_col] == test] + flags = test_rows[col] + if flags.isna().any(): + msg = ( + f"exclude_row column {col!r} has {int(flags.isna().sum())} NaN value(s) for turbine {test!r}; " + f"it must be boolean with no missing values (fill unknown rows with False explicitly)" + ) + raise ValueError(msg) + return flags.astype(bool).reindex(index, fill_value=False) + + def _write_outputs( + self, + mi: MethodInput, + *, + wide: pd.DataFrame, + stats: pd.DataFrame, + used: np.ndarray, + rho_base: float, + rho_up: float, + uplift: float, + sigma_overall: float, + n_refs: int, + timebase: pd.Timedelta, + per_bin: pd.DataFrame | None = None, + diagnostics: pd.DataFrame | None = None, + ) -> None: + """Write the data-stats CSV, the headline results CSV, the per-bin CSV and (optionally) the plots.""" + upgrade_start = toggle_upgrade_start(mi.upgrade_timing, wide.index) + last_dt = wide.index.max() + run_name = f"toggle_specialist_{mi.test_wtg}_{upgrade_start:%Y%m%d}_{last_dt:%Y%m%d}" + out_root = ( + Path(self.out_dir) if self.out_dir is not None else Path(tempfile.mkdtemp(prefix="toggle_specialist_")) + ) + run_dir = out_root / run_name + run_dir.mkdir(parents=True, exist_ok=True) + ts = pd.Timestamp.utcnow().strftime("%Y%m%d_%H%M%S_%f") + + stats.to_csv(run_dir / f"{run_name}_data_stats_{ts}.csv", index=False) + + used_base = int(stats.loc[stats["segment"] == "baseline", "n_used_timestamps"].iloc[0]) + used_up = int(stats.loc[stats["segment"] == "upgraded", "n_used_timestamps"].iloc[0]) + results = pd.DataFrame( + [ + { + "test_wtg": mi.test_wtg, + "mode": "toggle", + "n_turbines": wide.shape[1], + "n_refs": n_refs, + "ratio_baseline": rho_base, + "ratio_upgraded": rho_up, + "uplift_frc": uplift, + "uplift_sigma_frc": sigma_overall, + "block_hours": self.block_hours, + "n_resamples": self.n_resamples, + "n_used_timestamps_baseline": used_base, + "n_used_timestamps_upgraded": used_up, + "time_calculated": pd.Timestamp.utcnow(), + } + ] + ) + results.to_csv(run_dir / f"{run_name}_results_{ts}.csv", index=False) + + if per_bin is not None: + per_bin.to_csv(run_dir / f"{run_name}_by_power_bin_{ts}.csv", index=False) + if diagnostics is not None: + diagnostics.to_csv(run_dir / f"{run_name}_uncertainty_{ts}.csv", index=False) + + if self.save_plots: + _save_plots( + run_dir / "plots", + wide=wide, + mi=mi, + test=mi.test_wtg, + used=used, + timebase=timebase, + active_power_col=self.columns.active_power, + ) + if per_bin is not None: + _save_per_bin_plot( + run_dir / "plots" / stages.CONDITIONAL_UPLIFT / f"{mi.test_wtg}_per_bin_uplift.png", + per_bin=per_bin, + test=mi.test_wtg, + active_power_col=self.columns.active_power, + ) + self._write_shared_diagnostics(mi, run_dir=run_dir, wide=wide, timebase=timebase) + + def _write_shared_diagnostics( + self, mi: MethodInput, *, run_dir: Path, wide: pd.DataFrame, timebase: pd.Timedelta + ) -> None: + """Emit the shared cross-method diagnostics (coverage/curves/histograms) and the run config.""" + # ``wide`` (a pivot) drops all-NaN timestamps, so align the masks to the full unique index + # the DiagnosticContext uses (timestamps absent from ``wide`` are simply not used). + index = pd.DatetimeIndex(pd.unique(mi.scada_df.index)).sort_values() + test, refs = mi.test_wtg, [c for c in wide.columns if c != mi.test_wtg] + used_series = self._used_mask(mi, wide=wide, test=test, refs=refs, timebase=timebase) + used = used_series.reindex(index, fill_value=False).to_numpy() + treated = resolve_toggle(mi.upgrade_timing, index).upgraded.astype(bool) + ctx = DiagnosticContext( + run_dir=run_dir, + test_wtg=mi.test_wtg, + turbine_col=mi.turbine_col, + columns=self.columns, + scada_df=mi.scada_df, + treated_ts=treated, + used_ts=used, + timebase=timebase, + mode="toggle", + era5_df=None, + # the exclusion alone, so the plots show what the flag removed that downtime did not + excluded_ts=self._test_excluded(mi, test=test, index=index).to_numpy(), + ) + write_common_diagnostics(ctx) + params = { + "active_power_col": self.columns.active_power, + "availability_col": self.columns.availability, + } + write_run_config(ctx, method_name=self.name, method_params=params) + + +def _per_bin_counterfactual( + *, + label: npt.NDArray[np.float64], + test_pw: npt.NDArray[np.float64], + ref_total: npt.NDArray[np.float64], + baseline: npt.NDArray[np.bool_], + bins: list[float], +) -> npt.NDArray[np.float64]: + """Each row's counterfactual test power: its own bin's baseline ratio times its reference total. + + ``rho_base(b)`` is measured over the baseline rows of bin ``b``; every row (of either segment) then + takes the ``rho_base`` of the bin its ``label`` falls in. Rows in a bin with no baseline rows get + NaN, which is what makes an uncovered bin report NaN rather than an imputed value. + """ + assigned = pd.cut(label, bins=bins) + rho_by_bin = { + category: _rho(test_pw, ref_total, baseline & np.asarray(assigned == category)) + for category in assigned.categories + } + rho_row = np.asarray(pd.Series(assigned).map(rho_by_bin).astype(float)) + return rho_row * ref_total + + +def _cell_sigma(boot: BootstrapResult | None, cell: str) -> float: + """Return one cell's 1-sigma, or NaN when the bootstrap did not run or never saw that cell.""" + if boot is None or cell not in boot.cells: + return float("nan") + return boot.cells[cell].sigma + + +def _uncertainty_diagnostics( + boot: BootstrapResult | None, + *, + membership: dict[str, npt.NDArray[np.bool_]], + upgraded: npt.NDArray[np.bool_], + baseline: npt.NDArray[np.bool_], + used: npt.NDArray[np.bool_], +) -> pd.DataFrame: + """Per-cell account of how the uncertainty was reached, keyed by ``(condition, condition_bin)``. + + Carried through the harness seam uninterpreted, so an uncertainty model can be developed against + a saved sweep rather than by re-running one. Emitted even when the bootstrap did not run: the + counts are what explain why. Both counts are reported because a cell fails when either side of + its ratio runs out, and a single total would hide which. + """ + used_idx = np.flatnonzero(used) + up_used = upgraded[used_idx] + base_used = baseline[used_idx] + nan = float("nan") + rows = [] + for cell, member in membership.items(): + cell_boot = boot.cells[cell] if boot is not None and cell in boot.cells else None + rows.append( + { + "condition": _OVERALL if cell == _OVERALL else "power", + "condition_bin": cell, + "n_upgraded_records": int((member & up_used).sum()), + "n_baseline_records": int((member & base_used).sum()), + "n_blocks": boot.n_blocks if boot is not None else 0, + # Both components, not just the reported max: a blend rule can then be re-judged from + # a saved sweep rather than by re-running one. + "sigma_bootstrap": cell_boot.sigma_bootstrap if cell_boot is not None else nan, + "sigma_fallback": cell_boot.sigma_fallback if cell_boot is not None else nan, + "sigma_robust": cell_boot.sigma_robust if cell_boot is not None else nan, + "frac_resamples_finite": cell_boot.frac_resamples_finite if cell_boot is not None else nan, + } + ) + return pd.DataFrame(rows) + + +def _rho(test_pw: npt.NDArray[np.float64], ref_total: npt.NDArray[np.float64], mask: npt.NDArray[np.bool_]) -> float: + """Test-to-reference ratio over ``mask``: sum(test) / sum(ref_total). NaN if degenerate.""" + if not mask.any(): + return float("nan") + denom = ref_total[mask].sum() + if denom == 0: + return float("nan") + return float(test_pw[mask].sum() / denom) + + +def _rho_label(rho_base: float, rho_up: float) -> float: + """Return the test-to-reference ratio used to *label* bins: the mean of the two states. + + State-neutral by construction, so relabelling which state is the baseline cannot move a row + between bins. Still a campaign-level scalar, so the upgrade cannot move a row either. + """ + return 0.5 * (rho_base + rho_up) + + +def _segment_stats( + mi: MethodInput, + *, + wide: pd.DataFrame, + used: npt.NDArray[np.bool_], + toggle_rows: ToggleRowSets, + refs: list[str], + timebase: pd.Timedelta, + active_power_col: str, +) -> pd.DataFrame: + """Build the per-segment (all/baseline/upgraded) diagnostics table.""" + test = mi.test_wtg + test_pw = wide[test].to_numpy(dtype=float) + ref_total = wide[refs].sum(axis=1).to_numpy(dtype=float) + n_turbines = wide.shape[1] + timebase_hours = timebase / pd.Timedelta(hours=1) + + row_rows = resolve_toggle(mi.upgrade_timing, mi.scada_df.index) + row_power = mi.scada_df[active_power_col].to_numpy(dtype=float) + + ts_baseline = toggle_rows.campaign_baseline + row_baseline = row_rows.campaign_baseline + ts_masks = {"all": np.ones(len(wide), dtype=bool), "baseline": ts_baseline, "upgraded": toggle_rows.upgraded} + row_masks = {"all": np.ones(len(mi.scada_df), dtype=bool), "baseline": row_baseline, "upgraded": row_rows.upgraded} + + rows = [] + for segment in _SEGMENTS: + ts_mask = ts_masks[segment] + row_mask = row_masks[segment] + seg_ts = wide.index[ts_mask] + seg_used = used & ts_mask + n_used = int(seg_used.sum()) + + if len(seg_ts): + first, last = seg_ts.min(), seg_ts.max() + expected_ts = round((last - first) / timebase) + 1 + else: + first = last = pd.NaT + expected_ts = 0 + expected_rows = n_turbines * expected_ts + + n_rows = int(row_mask.sum()) + n_power_finite = int(np.isfinite(row_power[row_mask]).sum()) + + used_test = test_pw[seg_used] + used_ref = ref_total[seg_used] + rows.append( + { + "segment": segment, + "first_timestamp": first, + "last_timestamp": last, + "n_turbines": n_turbines, + "expected_timestamps": expected_ts, + "n_rows": n_rows, + "expected_rows": expected_rows, + "rows_data_coverage": n_rows / expected_rows if expected_rows else np.nan, + "n_power_finite_rows": n_power_finite, + "power_finite_coverage": n_power_finite / expected_rows if expected_rows else np.nan, + "n_used_timestamps": n_used, + "used_data_coverage": n_used / expected_ts if expected_ts else np.nan, + "used_test_mean_power_kw": float(used_test.mean()) if n_used else np.nan, + "used_test_mwh": float(used_test.sum()) * timebase_hours / 1000.0 if n_used else np.nan, + "used_ref_total_mean_power_kw": float(used_ref.mean()) if n_used else np.nan, + "used_ref_total_mwh": float(used_ref.sum()) * timebase_hours / 1000.0 if n_used else np.nan, + } + ) + return pd.DataFrame(rows) + + +def _daily_segment_ratio( + index: pd.DatetimeIndex, + test_pw: npt.NDArray[np.float64], + ref_total: npt.NDArray[np.float64], + seg_mask: npt.NDArray[np.bool_], +) -> pd.Series: + """Daily sum-based test/reference ratio (Sum test / Sum ref) over ``seg_mask`` rows; NaN on empty days. + + This matches the method's own ``rho`` definition (a ratio of sums, not a mean of per-timestamp + ratios), so the daily series fluctuates around the scalar ``rho`` the estimate uses instead of + blowing up on low-wind timestamps. + """ + test = pd.Series(np.where(seg_mask, test_pw, np.nan), index=index) + ref = pd.Series(np.where(seg_mask, ref_total, np.nan), index=index) + return test.resample("1D").sum(min_count=1) / ref.resample("1D").sum(min_count=1) + + +def _expected_per_day(index: pd.DatetimeIndex, timebase: pd.Timedelta) -> pd.Series: + """Daily count of timestamps the analysis timebase grid expects between the data's first and last.""" + grid = pd.date_range(index.min(), index.max(), freq=timebase) + return pd.Series(1.0, index=grid).resample("1D").sum() + + +def _daily_segment_coverage( + index: pd.DatetimeIndex, + used: npt.NDArray[np.bool_], + seg_mask: npt.NDArray[np.bool_], + expected_per_day: pd.Series, +) -> pd.Series: + """Daily used-data coverage in [0, 1], as a fraction of the day's expected timestamps. + + Numerator: complete-case timestamps (test and every reference finite) assigned to this segment + each day. Denominator: the day's expected timestamp count on the analysis timebase grid, which + is shared across segments. So the two segments' coverages sum to the day's overall complete-case + coverage, and under toggle each segment is capped near the duty cycle (~50%) of slots it can ever + occupy. NaN on days the grid does not reach. + """ + used_seg = pd.Series((used & seg_mask).astype(float), index=index) + daily_used = used_seg.resample("1D").sum() + return daily_used / expected_per_day.reindex(daily_used.index) + + +def _save_plots( + plots_dir: Path, + *, + wide: pd.DataFrame, + mi: MethodInput, + test: str, + used: np.ndarray, + timebase: pd.Timedelta, + active_power_col: str, +) -> None: + """Write the scatter, ratio-timeseries and used-coverage-timeseries diagnostic plots (by stage). + + ``used`` is the method's real downtime-filtered mask (test + every reference passing the + availability/finite filter), so the scatter shows only the rows the estimate actually uses. + The baseline is the strict campaign off-blocks the estimate used, so the plots never disagree + with the headline. + """ + refs = [c for c in wide.columns if c != test] + toggle_rows = resolve_toggle(mi.upgrade_timing, wide.index) + baseline_mask = toggle_rows.campaign_baseline + test_pw = wide[test].to_numpy(dtype=float) + ref_total = wide[refs].sum(axis=1).to_numpy(dtype=float) + upgrade_start = toggle_upgrade_start(mi.upgrade_timing, wide.index) + segments = ( + ("baseline", used & baseline_mask, "C0"), + ("upgraded", used & toggle_rows.upgraded, "C1"), + ) + + # 1) scatter of test vs reference-total power, baseline/upgraded coloured, with rho slopes. + fig, ax = plt.subplots(figsize=(7, 7)) + for label, seg, color in segments: + ax.scatter(ref_total[seg], test_pw[seg], s=8, alpha=0.4, color=color, label=label) + rho = _rho(test_pw, ref_total, seg) + if np.isfinite(rho) and seg.any(): + x_max = float(np.nanmax(ref_total[seg])) + ax.plot([0, x_max], [0, rho * x_max], color=color, linewidth=1.5) + ax.set_xlabel(f"sum of reference {active_power_col} [kW]") + ax.set_ylabel(f"{active_power_col} @ {test} [kW]") + ax.set_title(f"{test}: test vs reference-total power") + ax.grid(visible=True, alpha=0.3) + ax.legend() + fig.tight_layout() + _save(fig, plots_dir / stages.UPLIFT_INPUTS / f"{test}_scatter.png") + + # 2) daily sum-based test/ref ratio, one series per segment, with each segment's scalar rho overlaid. + fig, ax = plt.subplots(figsize=(10, 5)) + for label, seg, color in segments: + daily = _daily_segment_ratio(wide.index, test_pw, ref_total, seg) + ax.plot(daily.index.to_numpy(), daily.to_numpy(), marker=".", linewidth=0.8, color=color, label=label) + rho = _rho(test_pw, ref_total, seg) + span = wide.index[seg] + if np.isfinite(rho) and len(span): + ax.hlines(rho, span.min(), span.max(), color=color, linestyle="--", linewidth=1.5) + ax.axvline(upgrade_start, color="k", linestyle="--", label="upgrade start") + ax.set_xlabel("date") + ax.set_ylabel("test / reference-total ratio") + ax.set_title(f"{test}: daily test/reference ratio (dashed = rho used by estimate)") + ax.grid(visible=True, alpha=0.3) + ax.legend() + fig.tight_layout() + _save(fig, plots_dir / stages.UPLIFT_RESULTS / f"{test}_ratio_timeseries.png") + + # 3) daily used-data coverage as a fraction of the day's expected timestamps, one series per + # segment, so each segment is seen to receive its share (under toggle, ~50% each post-upgrade). + expected_per_day = _expected_per_day(wide.index, timebase) + fig, ax = plt.subplots(figsize=(10, 5)) + for label, _seg, color in segments: + seg_mask = baseline_mask if label == "baseline" else toggle_rows.upgraded + daily = _daily_segment_coverage(wide.index, used, seg_mask, expected_per_day) + ax.plot(daily.index.to_numpy(), daily.to_numpy(), marker=".", linewidth=0.8, color=color, label=label) + ax.axvline(upgrade_start, color="k", linestyle="--", label="upgrade start") + ax.set_ylim(0.0, 1.0) + ax.yaxis.set_major_formatter(PercentFormatter(xmax=1.0)) + ax.set_xlabel("date") + ax.set_ylabel("used-data coverage") + ax.set_title(f"{test}: daily used-data coverage (complete-case, % of expected timestamps)") + ax.grid(visible=True, alpha=0.3) + ax.legend() + fig.tight_layout() + _save(fig, plots_dir / stages.FILTER / f"{test}_coverage_timeseries.png") + + +def _save_per_bin_plot(path: Path, *, per_bin: pd.DataFrame, test: str, active_power_col: str) -> None: + """Plot the per-power-bin uplift with each bin's used-record count underneath. + + The record count is the point of the second panel: a per-bin uplift is only as trustworthy as the + data behind it, and the sparse bins are exactly where a reader must not over-read the top panel. + Empty bins are gaps, never plotted as zero. + """ + populated = per_bin["n_records"].to_numpy() > 0 + x = np.arange(len(per_bin)) + uplift = np.where(populated, per_bin["p50_uplift"].to_numpy() * 100.0, np.nan) + + fig, (ax_uplift, ax_n) = plt.subplots(2, 1, sharex=True, figsize=(9, 7), height_ratios=[2, 1]) + ax_uplift.plot(x, uplift, marker="o", color="C1") + ax_uplift.axhline(0.0, color="k", linewidth=0.8) + ax_uplift.set_ylabel("uplift [pp]") + ax_uplift.set_title(f"{test}: uplift by {active_power_col} bin") + ax_uplift.grid(visible=True, alpha=0.3) + + ax_n.bar(x, per_bin["n_records"].to_numpy(), color="C0", alpha=0.7) + ax_n.set_ylabel("used records") + ax_n.set_xlabel(f"{active_power_col} bin [kW] (predicted baseline)") + ax_n.set_xticks(x) + ax_n.set_xticklabels(per_bin["condition_bin"].astype(str), rotation=20, ha="right", fontsize=8) + ax_n.grid(visible=True, alpha=0.3) + fig.tight_layout() + _save(fig, path) + + +def _save(fig: plt.Figure, path: Path) -> None: + """Write a figure to ``path`` (creating its stage subfolder) and close it.""" + path.parent.mkdir(parents=True, exist_ok=True) + fig.savefig(path, dpi=150) + plt.close(fig) diff --git a/benchmarking/baselines/v0_binned.py b/benchmarking/baselines/v0_binned.py new file mode 100644 index 00000000..488fdbf6 --- /dev/null +++ b/benchmarking/baselines/v0_binned.py @@ -0,0 +1,205 @@ +"""The v0 binned power-curve method behind the harness ``Method`` seam. + +``V0BinnedMethod`` adapts a thin harness ``MethodInput`` to a full, faithful wind_up +pre/post power-performance run and returns its P50 uplift. For each campaign it renders a +per-campaign YAML and loads it with :meth:`WindUpConfig.from_yaml` (the standard way v0 +assessments are configured), then runs the real ``run_wind_up_analysis`` + ``combine_results``. + +Configuration is deliberately faithful: plots off, bootstrap untouched (its uncertainty is +required by ``combine_results``), ``ignore_turbine_anemometer_data`` / ``clip_rated_power_pp`` +at v0 defaults. Two choices are specific to this benchmarking exercise: + +* ``use_lt_distribution: False`` — we study *campaign* uplift (the injected ground truth), + not long-term uplift. +* ``combine_results(..., auto_choose_refs=False)`` — an analyst normally reviews/chooses refs. + +With ``years_offset_for_pre_period: 1`` and ``years_for_{lt_distribution,detrend}: 1``, +``from_yaml`` derives a seasonally-matched pre period (the post window shifted back one year) +and a one-year detrend window; the harness only provides ~12 months of pre data, so the pre +and detrend windows are constrained accordingly — the campaign-length degradation the harness +exists to measure. + +Both prepost and toggle inputs are supported. A ``ToggleSchedule`` selects wind_up's native +toggle assessment: the config carries a ``toggle:`` block (``detrend_data_selection: +use_toggle_off_data``, settling filter 0) instead of the prepost offset fields, and the on/off +signal is supplied as a ``toggle_df`` built from the schedule (see ``_build_toggle_df``). For +toggle, the power-performance split uses campaign data only (``analysis_first`` = +``upgrade_first``), while pre-processing/detrend/long-term steps still use pre-campaign data. +""" + +from __future__ import annotations + +import tempfile +from dataclasses import dataclass +from pathlib import Path +from typing import TYPE_CHECKING + +import pandas as pd + +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.harness.toggle import build_toggle_df, is_toggle, toggle_upgrade_start +from benchmarking.synthetic.sources.hill_of_towie import long_to_wind_up_format +from wind_up.combine_results import combine_results +from wind_up.interface import AssessmentInputs +from wind_up.main_analysis import run_wind_up_analysis +from wind_up.models import PlotConfig, WindUpConfig + +if TYPE_CHECKING: + from benchmarking.baselines.hot_context import HotV0Context + +_CAMPAIGN_YAML_TEMPLATE = """\ +assessment_name: {assessment_name} +test_wtgs: + - {test_wtg} +ref_wtgs: +{ref_lines} +upgrade_first_dt_utc_start: {upgrade} +analysis_last_dt_utc_start: {analysis_last} +years_offset_for_pre_period: 1 +years_for_lt_distribution: 1 +years_for_detrend: 1 +use_lt_distribution: false +ws_bin_width: {ws_bin_width} +reanalysis_method: {reanalysis_method} +optimize_northing_corrections: false +northing_corrections_utc: !include {northing_yaml} +asset: !include {asset_yaml} +""" + +# Toggle variant: wind_up's native toggle assessment. ``analysis_first`` is derived as the +# toggle campaign start (``upgrade_first``), so the on/off power-performance split uses only +# campaign data; long-term/detrend windows still reach back into pre-campaign data. The toggle +# signal is supplied directly as a ``toggle_df`` (see ``_build_toggle_df``), so ``toggle_filename`` +# is a never-read placeholder. The settling filter is 0 because the synthetic toggle blocks are +# short and have no real settling transient. +_TOGGLE_YAML_TEMPLATE = """\ +assessment_name: {assessment_name} +test_wtgs: + - {test_wtg} +ref_wtgs: +{ref_lines} +upgrade_first_dt_utc_start: {upgrade} +analysis_last_dt_utc_start: {analysis_last} +years_for_lt_distribution: 1 +years_for_detrend: 1 +use_lt_distribution: false +ws_bin_width: {ws_bin_width} +reanalysis_method: {reanalysis_method} +optimize_northing_corrections: false +toggle: + toggle_file_per_turbine: false + toggle_filename: not_used.parquet + detrend_data_selection: use_toggle_off_data + toggle_change_settling_filter_seconds: 0 +northing_corrections_utc: !include {northing_yaml} +asset: !include {asset_yaml} +""" + + +def _subset_turbines(scada_df: pd.DataFrame, turbine_col: str) -> list[str]: + """Return the sorted unique turbine names present in ``scada_df``.""" + return sorted(scada_df[turbine_col].unique().tolist()) + + +def _extract_p50(tdf: pd.DataFrame, test_wtg: str) -> float: + """Return the combined P50 uplift fraction for the (non-reference) test turbine.""" + row = tdf.loc[(tdf["test_wtg"] == test_wtg) & (~tdf["is_ref"])] + if len(row) != 1: + msg = f"expected exactly one non-ref combined result row for test_wtg {test_wtg!r}, found {len(row)}" + raise ValueError(msg) + return float(row["p50_uplift"].iloc[0]) + + +@dataclass +class V0BinnedMethod: + """Pluggable v0 binned power-curve baseline. + + :param context: the shared HoT source-context (metadata, reanalysis, vendored asset/northing) + :param name: method name shown in the leaderboard + :param ws_bin_width: power-curve wind-speed bin width in m/s + :param reanalysis_method: wind_up reanalysis-node selection method + :param scratch_dir: where per-campaign YAML + wind_up output go; a temp dir when ``None`` + :param save_plots: if True, save wind_up's per-campaign plots under ``/plots`` (each + campaign has its own unique out dir); off by default since plots are slow and unused for + scoring, but useful for manually inspecting a run + """ + + context: HotV0Context + name: str = "v0_binned" + ws_bin_width: float = 1.0 + reanalysis_method: str = "node_with_best_ws_corr" + scratch_dir: Path | None = None + save_plots: bool = False + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Run a faithful v0 power-performance analysis (prepost or toggle) and return its P50 uplift.""" + cfg = self._build_config(mi) + plot_cfg = PlotConfig(show_plots=False, save_plots=self.save_plots, plots_dir=cfg.out_dir / "plots") + from_cfg_kwargs: dict = { + "cfg": cfg, + "plot_cfg": plot_cfg, + # The harness hands us source-native SCADA; v0 needs wind-up format, so convert here + # (the v0 baseline is the only place that knows v0's column names). The conversion + # returns a fresh frame, so there is no in-place mutation of the harness's slice. + "scada_df": long_to_wind_up_format(mi.scada_df), + "metadata_df": self.context.metadata_df, + "reanalysis_datasets": self.context.reanalysis_datasets, + "cache_dir": None, + } + if is_toggle(mi.upgrade_timing): + # A ToggleSchedule is turned into the canonical toggle_df; an explicit toggle_df passes through. + from_cfg_kwargs["toggle_df"] = ( + mi.upgrade_timing + if isinstance(mi.upgrade_timing, pd.DataFrame) + else build_toggle_df(mi.scada_df.index, mi.upgrade_timing) + ) + inputs = AssessmentInputs.from_cfg(**from_cfg_kwargs) + trdf = run_wind_up_analysis(inputs) + tdf = combine_results(trdf, auto_choose_refs=False, plot_config=None) + tdf.to_csv( + cfg.out_dir + / f"{cfg.assessment_name}_combined_results_{pd.Timestamp.utcnow().strftime('%Y%m%d_%H%M%S')}.csv" + ) + return MethodOutput(p50_overall=_extract_p50(tdf, mi.test_wtg)) + + def _build_config(self, mi: MethodInput) -> WindUpConfig: + """Render and load the per-campaign WindUpConfig, with the asset filtered to the subset.""" + subset = _subset_turbines(mi.scada_df, mi.turbine_col) + refs = [t for t in subset if t != mi.test_wtg] + if not refs: + msg = ( + f"no reference turbines available for test_wtg {mi.test_wtg!r}: scada_df contains only " + f"{subset}. The v0 binned method needs at least one reference turbine." + ) + raise ValueError(msg) + toggle_mode = is_toggle(mi.upgrade_timing) + upgrade = toggle_upgrade_start(mi.upgrade_timing, mi.scada_df.index) + analysis_last = pd.Timestamp(mi.scada_df.index.max()) + assessment_name = f"v0_{mi.test_wtg}_{upgrade:%Y%m%d}_{analysis_last:%Y%m%d}" + + scratch = Path(self.scratch_dir) if self.scratch_dir is not None else Path(tempfile.mkdtemp(prefix="v0_")) + scratch.mkdir(parents=True, exist_ok=True) + template = _TOGGLE_YAML_TEMPLATE if toggle_mode else _CAMPAIGN_YAML_TEMPLATE + yaml_text = template.format( + assessment_name=assessment_name, + test_wtg=mi.test_wtg, + ref_lines="\n".join(f" - {r}" for r in refs), + upgrade=upgrade.strftime("%Y-%m-%d %H:%M:%S"), + analysis_last=analysis_last.strftime("%Y-%m-%d %H:%M:%S"), + ws_bin_width=self.ws_bin_width, + reanalysis_method=self.reanalysis_method, + northing_yaml=self.context.northing_yaml.as_posix(), + asset_yaml=self.context.asset_yaml.as_posix(), + ) + yaml_path = scratch / f"{assessment_name}.yaml" + yaml_path.write_text(yaml_text) + + cfg = WindUpConfig.from_yaml(yaml_path) + cfg.out_dir = scratch / assessment_name + cfg.out_dir.mkdir(parents=True, exist_ok=True) + # The asset YAML lists all 21 HoT turbines, but only the subset has SCADA here. Restrict + # it so wind farm coverage (e.g. reanalysis correlation) is computed over the right count. + cfg.asset.wtgs = [w for w in cfg.asset.wtgs if w.name in subset] + # similar subsetting logic for northing_corrections_utc + cfg.northing_corrections_utc = [n for n in cfg.northing_corrections_utc if n[0] in subset] + return cfg diff --git a/benchmarking/diagnostics/__init__.py b/benchmarking/diagnostics/__init__.py new file mode 100644 index 00000000..55ce5ec3 --- /dev/null +++ b/benchmarking/diagnostics/__init__.py @@ -0,0 +1,27 @@ +"""Shared per-run diagnostics for the v1 benchmarking methods. + +A method adapts its internals to a :class:`~benchmarking.diagnostics.context.DiagnosticContext` +and calls :func:`~benchmarking.diagnostics.common.write_common_diagnostics` (plus +:func:`~benchmarking.diagnostics.config_dump.write_run_config`) to emit a consistent, v0-grade +set of diagnostic plots and a run-config file. Project plotting conventions live in +:mod:`~benchmarking.diagnostics.style` (grid on by default) and +:mod:`~benchmarking.diagnostics.density` (density-coloured scatter). +""" + +from __future__ import annotations + +from benchmarking.diagnostics.common import write_common_diagnostics +from benchmarking.diagnostics.config_dump import write_run_config +from benchmarking.diagnostics.context import DiagnosticContext, infer_timebase +from benchmarking.diagnostics.density import density_scatter +from benchmarking.diagnostics.style import apply_grid, save_fig + +__all__ = [ + "DiagnosticContext", + "apply_grid", + "density_scatter", + "infer_timebase", + "save_fig", + "write_common_diagnostics", + "write_run_config", +] diff --git a/benchmarking/diagnostics/common.py b/benchmarking/diagnostics/common.py new file mode 100644 index 00000000..d60f7af3 --- /dev/null +++ b/benchmarking/diagnostics/common.py @@ -0,0 +1,81 @@ +"""The single entry point a method calls to emit the shared cross-method diagnostics. + +:func:`write_common_diagnostics` runs every shared plot for a :class:`DiagnosticContext`. Each +plot is independent and guarded: one failing (e.g. a degenerate segment) logs and is skipped +rather than killing the rest of an unattended inspection run. Plots that need an absent signal +return ``None`` and are simply not written. +""" + +from __future__ import annotations + +import logging +from typing import TYPE_CHECKING + +from benchmarking.diagnostics.coverage import ( + plot_excluded_fraction, + plot_filter_coverage, + plot_input_coverage, + plot_input_timeline, +) +from benchmarking.diagnostics.curves import ( + plot_curves_by_upgrade, + plot_ops_curves, + plot_ops_curves_excluded, + plot_ops_curves_kept, + plot_power_factor, + plot_reactive_vs_active, +) +from benchmarking.diagnostics.histograms import plot_condition_histograms +from benchmarking.diagnostics.northing import plot_northing_error + +if TYPE_CHECKING: + from collections.abc import Callable + from pathlib import Path + + from benchmarking.diagnostics.context import DiagnosticContext + +logger = logging.getLogger(__name__) + +# Every shared plot, in a sensible reading order. Each takes the context and returns a path +# (or None when it has nothing to draw for this source). +_PLOTS: tuple[Callable[[DiagnosticContext], Path | None], ...] = ( + plot_input_timeline, + plot_input_coverage, + plot_filter_coverage, + plot_excluded_fraction, + plot_condition_histograms, + plot_ops_curves, + plot_ops_curves_kept, + plot_ops_curves_excluded, + plot_curves_by_upgrade, + plot_reactive_vs_active, + plot_power_factor, + plot_northing_error, +) + + +def write_common_diagnostics(ctx: DiagnosticContext) -> list[Path]: + """Write every shared diagnostic plot for ``ctx``; return the paths actually written.""" + n = len(ctx.index) + excluded_n = None if ctx.excluded_ts is None else len(ctx.excluded_ts) + if len(ctx.treated_ts) != n or len(ctx.used_ts) != n or excluded_n not in (None, n): + logger.error( + "diagnostic masks misaligned with the index (index=%d, treated=%d, used=%d, excluded=%s);" + " skipping diagnostics", + n, + len(ctx.treated_ts), + len(ctx.used_ts), + excluded_n, + ) + return [] + ctx.plots_dir.mkdir(parents=True, exist_ok=True) + written: list[Path] = [] + for plot in _PLOTS: + try: + path = plot(ctx) + except Exception: + logger.exception("diagnostic plot %s failed for %s", plot.__name__, ctx.test_wtg) + continue + if path is not None: + written.append(path) + return written diff --git a/benchmarking/diagnostics/config_dump.py b/benchmarking/diagnostics/config_dump.py new file mode 100644 index 00000000..1180aace --- /dev/null +++ b/benchmarking/diagnostics/config_dump.py @@ -0,0 +1,125 @@ +"""Write a human-readable record of what a diagnostic run actually received. + +v0 drops a ``config_*.json`` per run and it is very handy for confirming the inputs after the +fact (feedback 2026-06-26). The benchmarking methods get the same, as YAML: the method and its +parameters, the run's window / turbines / timebase, the columns in play, and the git commit. +""" + +from __future__ import annotations + +import subprocess +from pathlib import Path +from typing import TYPE_CHECKING, Any + +import numpy as np +import pandas as pd +import yaml + +if TYPE_CHECKING: + from benchmarking.diagnostics.context import DiagnosticContext + +_REPO_DIR = Path(__file__).resolve().parent + + +def _git_commit() -> str: + """Return the current commit (``-dirty`` suffix if the tree has uncommitted changes), or ``unknown``.""" + try: + commit = subprocess.run( + ["git", "rev-parse", "HEAD"], # noqa: S607 + cwd=_REPO_DIR, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + dirty = subprocess.run( + ["git", "status", "--porcelain"], # noqa: S607 + cwd=_REPO_DIR, + capture_output=True, + text=True, + check=True, + ).stdout.strip() + except (subprocess.SubprocessError, OSError): + return "unknown" + return f"{commit}-dirty" if dirty else commit + + +def _diagnostic_columns(ctx: DiagnosticContext) -> dict[str, str | None]: + """Return the source-native column names in play (so a reader knows what each plot read).""" + cols = ctx.columns + return { + "turbine": cols.turbine, + "active_power": cols.active_power, + "wind_speed": cols.wind_speed, + "wind_speed_sd": cols.wind_speed_sd, + "gen_rpm": cols.gen_rpm, + "pitch": cols.pitch, + "reactive_power": cols.reactive_power, + "nacelle_position": cols.nacelle_position, + "ambient_temp": cols.ambient_temp, + "availability": cols.availability, + } + + +def write_run_config( + ctx: DiagnosticContext, + *, + method_name: str, + method_params: dict[str, Any], + extra: dict[str, Any] | None = None, +) -> Path: + """Write ``/config_.yaml`` describing the run; return its path. + + :param method_name: the method's leaderboard name + :param method_params: the method's configuration (plain, YAML-serialisable values) + :param extra: optional method-specific extras to record (e.g. ERA5 lag, n_folds) + """ + index = ctx.index + used = np.asarray(ctx.used_ts, dtype=bool) + treated = np.asarray(ctx.treated_ts, dtype=bool) + record: dict[str, Any] = { + "method": method_name, + "method_params": {k: _yamlable(v) for k, v in method_params.items()}, + "mode": ctx.mode, + "test_wtg": ctx.test_wtg, + "references": ctx.references(), + "n_turbines": int(ctx.scada_df[ctx.turbine_col].nunique()), + "first_timestamp": _yamlable(index.min()) if len(index) else None, + "last_timestamp": _yamlable(index.max()) if len(index) else None, + "timebase_seconds": ctx.timebase.total_seconds(), + "n_timestamps": len(index), + "n_used_timestamps": int(used.sum()), + "n_used_upgraded": int((used & treated).sum()), + "n_used_baseline": int((used & ~treated).sum()), + "has_era5": ctx.era5_df is not None, + "columns": _diagnostic_columns(ctx), + "git_commit": _git_commit(), + } + if extra: + record["extra"] = {k: _yamlable(v) for k, v in extra.items()} + + ctx.run_dir.mkdir(parents=True, exist_ok=True) + path = ctx.run_dir / f"config_{ctx.run_dir.name}.yaml" + path.write_text(yaml.safe_dump(record, sort_keys=False, default_flow_style=False)) + return path + + +def _yamlable(value: Any) -> Any: # noqa: ANN401 + """Convert numpy/pandas scalars and containers to plain Python so ``yaml.safe_dump`` accepts them.""" + if isinstance(value, dict): + return {k: _yamlable(v) for k, v in value.items()} + if isinstance(value, (list, tuple)): + return [_yamlable(v) for v in value] + return _yamlable_scalar(value) + + +def _yamlable_scalar(value: Any) -> Any: # noqa: ANN401 + """Convert a single numpy/pandas scalar (or Path) to a plain YAML-serialisable value.""" + if isinstance(value, pd.Timestamp): + return value.isoformat() + if isinstance(value, pd.Timedelta): + return value.total_seconds() + if isinstance(value, np.generic): + return value.item() + if isinstance(value, Path): + return str(value) + return value diff --git a/benchmarking/diagnostics/context.py b/benchmarking/diagnostics/context.py new file mode 100644 index 00000000..62c8bb15 --- /dev/null +++ b/benchmarking/diagnostics/context.py @@ -0,0 +1,139 @@ +"""The shared inputs the cross-method diagnostics draw from. + +A method adapts its internals to a :class:`DiagnosticContext` (the test turbine, the long SCADA +slice, per-timestamp treatment/used masks, the timebase, optional aligned ERA5) and the shared +plotting functions take it from there. This keeps the plot code method-agnostic: it knows +nothing about R-learner folds or the naive ratio, only the common picture of "which turbine, +which rows, used or not, baseline or upgraded". + +The few computed views the plots need (the unique index, the test turbine's rows aligned to it, +a wide per-turbine pivot, a reference-mean signal) live here so the plotting modules stay lean. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd + +if TYPE_CHECKING: + from pathlib import Path + + import numpy.typing as npt + + from benchmarking.synthetic import ColumnSchema + +# Column names the optional ``era5_df`` is expected to carry (the R-learner's ERA5 sync output). +ERA5_WS_COL = "era5_ws" +ERA5_WD_COL = "era5_wd" + +_MIN_POINTS_FOR_TIMEBASE = 2 + + +def infer_timebase(index: pd.DatetimeIndex) -> pd.Timedelta: + """Infer the analysis timebase as the median spacing of the sorted unique timestamps.""" + unique = pd.DatetimeIndex(pd.unique(index)).sort_values() + if len(unique) < _MIN_POINTS_FOR_TIMEBASE: + return pd.Timedelta(minutes=10) + return pd.Timedelta(np.median(np.diff(unique.to_numpy()))) + + +@dataclass +class DiagnosticContext: + """Method-agnostic inputs for the shared per-run diagnostics. + + :param run_dir: the method's per-run output folder (plots land in ``run_dir/"plots"``) + :param test_wtg: the test turbine name + :param turbine_col: the long frame's turbine-identifier column + :param columns: the source-native column schema (diagnostic roles may be ``None``) + :param scada_df: the ``MethodInput`` SCADA slice (long format, all subset turbines) + :param treated_ts: treatment flag (0/1) per unique timestamp, in sorted-index order + :param used_ts: "used by the method" flag per unique timestamp (the test turbine's kept rows) + :param timebase: the analysis timebase + :param mode: ``"prepost"`` or ``"toggle"`` + :param era5_df: optional ERA5 aligned to the unique index (columns :data:`ERA5_WS_COL` / + :data:`ERA5_WD_COL`) + :param excluded_ts: optional per-timestamp ``ColumnSchema.exclude_row`` mask (``None`` for a + method with no exclusion concept). Separate from ``used_ts``, which also folds in downtime. + """ + + run_dir: Path + test_wtg: str + turbine_col: str + columns: ColumnSchema + scada_df: pd.DataFrame + treated_ts: npt.NDArray[np.bool_] + used_ts: npt.NDArray[np.bool_] + timebase: pd.Timedelta + mode: str + era5_df: pd.DataFrame | None = None + excluded_ts: npt.NDArray[np.bool_] | None = None + + @property + def index(self) -> pd.DatetimeIndex: + """The unique, sorted analysis timestamps (one entry per timebase slot).""" + return pd.DatetimeIndex(pd.unique(self.scada_df.index)).sort_values() + + @property + def plots_dir(self) -> Path: + """Root folder the diagnostic plots are written under (stage subfolders live here).""" + return self.run_dir / "plots" + + def stage_dir(self, stage: str) -> Path: + """Return (and create) the plots subfolder for an analysis ``stage`` (see :mod:`stages`).""" + path = self.plots_dir / stage + path.mkdir(parents=True, exist_ok=True) + return path + + @property + def baseline_ts(self) -> npt.NDArray[np.bool_]: + """Per-timestamp mask of baseline (un-upgraded) slots.""" + return ~np.asarray(self.treated_ts, dtype=bool) + + @property + def upgraded_ts(self) -> npt.NDArray[np.bool_]: + """Per-timestamp mask of upgraded slots.""" + return np.asarray(self.treated_ts, dtype=bool) + + def excluded_mask(self) -> npt.NDArray[np.bool_] | None: + """Caller-flagged exclusions as a bool mask; ``None`` when unset or empty (nothing to draw).""" + if self.excluded_ts is None: + return None + mask = np.asarray(self.excluded_ts, dtype=bool) + return mask if mask.any() else None + + def references(self) -> list[str]: + """Sorted reference turbine names (every turbine present except the test turbine).""" + return sorted(t for t in self.scada_df[self.turbine_col].unique() if t != self.test_wtg) + + def has_column(self, col: str | None) -> bool: + """Return True if ``col`` is named (not ``None``) and present in the SCADA frame.""" + return col is not None and col in self.scada_df.columns + + def turbine_series(self, turbine: str, col: str | None) -> pd.Series: + """Return a turbine's ``col`` aligned to the unique index (NaN where missing/absent).""" + if col is None or col not in self.scada_df.columns: + return pd.Series(np.nan, index=self.index) + rows = self.scada_df[self.scada_df[self.turbine_col] == turbine] + series = pd.Series(rows[col].to_numpy(dtype=float), index=pd.DatetimeIndex(rows.index)) + return series[~series.index.duplicated()].reindex(self.index) + + def test_series(self, col: str | None) -> pd.Series: + """Return the test turbine's ``col`` aligned to the unique index.""" + return self.turbine_series(self.test_wtg, col) + + def reference_mean(self, col: str | None) -> pd.Series: + """Mean of ``col`` across reference turbines, aligned to the unique index.""" + refs = self.references() + if not refs: + return pd.Series(np.nan, index=self.index) + frame = pd.concat([self.turbine_series(r, col) for r in refs], axis=1) + return frame.mean(axis=1) + + def wide(self, col: str) -> pd.DataFrame: + """Timestamp x turbine pivot of ``col`` (NaN where missing), on the unique index.""" + tmp = self.scada_df[[self.turbine_col, col]].copy() + tmp["_ts"] = self.scada_df.index + return tmp.pivot_table(index="_ts", columns=self.turbine_col, values=col, aggfunc="first").reindex(self.index) diff --git a/benchmarking/diagnostics/coverage.py b/benchmarking/diagnostics/coverage.py new file mode 100644 index 00000000..e921ac2b --- /dev/null +++ b/benchmarking/diagnostics/coverage.py @@ -0,0 +1,176 @@ +"""Data-coverage diagnostics: where, in time, each turbine (and ERA5) actually has data. + +Three views (feedback 2026-06-26): a v0-style **window timeline** (data-present periods per +turbine + ERA5, with baseline/upgraded bands), a **raw coverage** line-plot (% present over time, +before any filtering), and a **filter coverage** line-plot (test turbine, raw vs used, so the +effect of the row filter is visible). Heatmaps were dropped — they were hard to read. + +Wind speed / power columns are referred to by their original source-native names. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib.dates as mdates +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +from matplotlib.patches import Patch +from matplotlib.ticker import PercentFormatter + +from benchmarking.diagnostics import stages +from benchmarking.diagnostics.context import ERA5_WS_COL +from benchmarking.diagnostics.style import apply_grid, save_fig +from benchmarking.diagnostics.timeaxis import BASELINE_COLOR, UPGRADED_COLOR, shade_segments + +if TYPE_CHECKING: + from pathlib import Path + + from benchmarking.diagnostics.context import DiagnosticContext + +_COVERAGE_BUCKET = "7D" # weekly buckets keep a multi-year window legible +_WEEKLY_BUCKET_FROM = pd.Timedelta(days=60) # below this a weekly timeline is too few points to read + + +def _present_runs(index: pd.DatetimeIndex, present: np.ndarray) -> list[tuple[float, float]]: + """Contiguous (start, width) spans in matplotlib date units where ``present`` is True.""" + runs: list[tuple[float, float]] = [] + if not present.any(): + return runs + x = mdates.date2num(index.to_pydatetime()) + edges = np.flatnonzero(np.diff(np.concatenate([[0], present.astype(int), [0]]))) + for start_i, end_i in zip(edges[::2], edges[1::2], strict=True): + x0 = x[start_i] + x1 = x[min(end_i, len(x) - 1)] + runs.append((x0, max(x1 - x0, 1e-6))) + return runs + + +def plot_input_timeline(ctx: DiagnosticContext) -> Path: + """Window timeline: data-present periods per turbine (+ ERA5), with baseline/upgraded bands.""" + rows: list[tuple[str, np.ndarray]] = [] + for turbine in [ctx.test_wtg, *ctx.references()]: + label = f"{turbine}{' (test)' if turbine == ctx.test_wtg else ''}: {ctx.columns.active_power}" + rows.append((label, ctx.turbine_series(turbine, ctx.columns.active_power).notna().to_numpy())) + if ctx.era5_df is not None and ERA5_WS_COL in ctx.era5_df.columns: + rows.append((f"ERA5: {ERA5_WS_COL}", ctx.era5_df[ERA5_WS_COL].reindex(ctx.index).notna().to_numpy())) + + fig, ax = plt.subplots(figsize=(13, 1.2 + 0.5 * len(rows))) + shade_segments(ax, ctx) + for y, (label, present) in enumerate(rows): # noqa: B007 - y is the row position + ax.broken_barh(_present_runs(ctx.index, present), (y + 0.6, 0.8), facecolors="C0") + ax.set_yticks(np.arange(len(rows)) + 1.0) + ax.set_yticklabels([label for label, _ in rows]) + ax.set_ylim(0.5, len(rows) + 0.5) + ax.xaxis_date() + ax.xaxis.set_major_formatter(mdates.DateFormatter("%Y-%m")) + ax.set_xlabel("date") + ax.set_title(f"{ctx.test_wtg}: input data timeline (bars = data present)") + legend = [ + Patch(color=BASELINE_COLOR, alpha=0.4, label="baseline"), + Patch(color=UPGRADED_COLOR, alpha=0.4, label="upgraded"), + ] + ax.legend(handles=legend, loc="lower right") + apply_grid(ax) + path = ctx.stage_dir(stages.INPUTS) / "input_data_timeline.png" + save_fig(fig, path) + return path + + +def _weekly_coverage(present: pd.Series, *, timebase: pd.Timedelta) -> pd.Series: + """Weekly fraction (%) of expected timebase slots for which ``present`` is True.""" + expected = pd.Timedelta(_COVERAGE_BUCKET) / timebase + return 100.0 * present.astype(float).resample(_COVERAGE_BUCKET).sum() / expected + + +def plot_input_coverage(ctx: DiagnosticContext) -> Path: + """Weekly %-coverage line per turbine's power (+ ERA5 wind speed), before any filtering.""" + fig, ax = plt.subplots(figsize=(12, 6)) + shade_segments(ax, ctx) + for turbine in [ctx.test_wtg, *ctx.references()]: + present = pd.Series(ctx.turbine_series(turbine, ctx.columns.active_power).notna().to_numpy(), index=ctx.index) + weekly = _weekly_coverage(present, timebase=ctx.timebase) + label = f"{turbine}{' (test)' if turbine == ctx.test_wtg else ''}" + ax.plot(weekly.index.to_numpy(), weekly.to_numpy(), linewidth=1.0, label=label) + if ctx.era5_df is not None and ERA5_WS_COL in ctx.era5_df.columns: + era5_present = pd.Series(ctx.era5_df[ERA5_WS_COL].reindex(ctx.index).notna().to_numpy(), index=ctx.index) + weekly = _weekly_coverage(era5_present, timebase=ctx.timebase) + ax.plot(weekly.index.to_numpy(), weekly.to_numpy(), linewidth=1.0, linestyle="--", label="ERA5") + ax.set_ylim(0, 105) + ax.set_xlabel("date") + ax.set_ylabel(f"weekly {ctx.columns.active_power} coverage [%]") + ax.set_title(f"{ctx.test_wtg}: input data coverage (before filtering)") + apply_grid(ax) + ax.legend(ncol=2, fontsize="small") + path = ctx.stage_dir(stages.INPUTS) / "input_data_coverage.png" + save_fig(fig, path) + return path + + +def plot_filter_coverage(ctx: DiagnosticContext) -> Path: + """Weekly %-coverage of the test turbine: raw present vs used (after the row filter).""" + raw = pd.Series(ctx.test_series(ctx.columns.active_power).notna().to_numpy(), index=ctx.index) + used = pd.Series(np.asarray(ctx.used_ts, dtype=bool), index=ctx.index) + fig, ax = plt.subplots(figsize=(12, 6)) + shade_segments(ax, ctx) + ax.plot( + _weekly_coverage(raw, timebase=ctx.timebase).index.to_numpy(), + _weekly_coverage(raw, timebase=ctx.timebase).to_numpy(), + linewidth=1.2, + color="C0", + label=f"{ctx.columns.active_power} present (raw)", + ) + ax.plot( + _weekly_coverage(used, timebase=ctx.timebase).index.to_numpy(), + _weekly_coverage(used, timebase=ctx.timebase).to_numpy(), + linewidth=1.2, + color="C3", + label="used (after filter)", + ) + ax.set_ylim(0, 105) + ax.yaxis.set_major_formatter(PercentFormatter()) + ax.set_xlabel("date") + ax.set_ylabel("weekly coverage [%]") + ax.set_title(f"{ctx.test_wtg}: data coverage before vs after the row filter") + apply_grid(ax) + ax.legend() + path = ctx.stage_dir(stages.FILTER) / "filter_coverage.png" + save_fig(fig, path) + return path + + +def exclusion_bucket(index: pd.DatetimeIndex) -> str: + """Resample rule for the exclusion timeline: a short campaign is one weekly dot, so bucket daily.""" + span = index.max() - index.min() if len(index) else pd.Timedelta(0) + return _COVERAGE_BUCKET if span > _WEEKLY_BUCKET_FROM else "1D" + + +def plot_excluded_fraction(ctx: DiagnosticContext) -> Path | None: + """%-share of the test turbine's timestamps removed by the caller's exclusion flag, over time. + + A flag firing in concentrated blocks can coincide with a toggle state and bias the ratio, which + the curves cannot show. ``None`` when the method has no exclusion mask or excluded nothing. + """ + excluded = ctx.excluded_mask() + if excluded is None: + return None + bucket = exclusion_bucket(ctx.index) + flagged = pd.Series(excluded, index=ctx.index) + share = 100.0 * flagged.astype(float).resample(bucket).mean() + + fig, ax = plt.subplots(figsize=(12, 6)) + shade_segments(ax, ctx) + ax.plot(share.index.to_numpy(), share.to_numpy(), linewidth=1.2, color="C3", marker=".") + ax.set_ylim(0, 105) + ax.yaxis.set_major_formatter(PercentFormatter()) + ax.set_xlabel("date") + ax.set_ylabel(f"timestamps excluded per {bucket} [%]") + ax.set_title( + f"{ctx.test_wtg}: rows excluded by the caller's flag " + f"({excluded.sum()} of {len(excluded)}, {100.0 * excluded.mean():.1f}%)" + ) + apply_grid(ax) + path = ctx.stage_dir(stages.FILTER) / "excluded_row_fraction.png" + save_fig(fig, path) + return path diff --git a/benchmarking/diagnostics/curves.py b/benchmarking/diagnostics/curves.py new file mode 100644 index 00000000..4a6059bc --- /dev/null +++ b/benchmarking/diagnostics/curves.py @@ -0,0 +1,236 @@ +"""Operating-curve and reactive-power diagnostics for the test turbine (feedback 2026-06-26). + +These are SCADA normal-operation diagnostics, so they use the **turbine's own** wind speed (the +operationally meaningful signal), not a reference mean — even though own wind speed is never a +*model feature* (it is post-treatment, design-note §3). Columns are labelled by their original +source-native names. + +* :func:`plot_ops_curves` — a 2x3 figure (power curve; pitch/rpm vs power; pitch/rpm vs wind + speed) coloured kept vs removed, so it is both the operating-curve view and the filter check + (stage: filter). +* :func:`plot_ops_curves_excluded` — the same figure coloured kept vs caller-excluded (stage: filter). +* :func:`plot_curves_by_upgrade` — pitch/rpm/power vs wind speed split baseline vs upgraded + (stage: uplift inputs). +* :func:`plot_reactive_vs_active` / :func:`plot_power_factor` — reactive-power behaviour, per + turbine (stage: inputs). +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np + +from benchmarking.diagnostics import stages +from benchmarking.diagnostics.style import apply_grid, save_fig +from benchmarking.diagnostics.timeaxis import shade_segments + +if TYPE_CHECKING: + from pathlib import Path + + import pandas as pd + + from benchmarking.diagnostics.context import DiagnosticContext + +_MIN_FINITE = 2 +_PF_BUCKET = "MS" # monthly buckets keep the power-factor / northing timelines smooth + +# A plotting "segment": (legend label, boolean row mask, colour). +Segments = list[tuple[str, np.ndarray, str]] + + +def _own_ws(ctx: DiagnosticContext) -> np.ndarray | None: + """Return the test turbine's own wind speed, or None if too little of it to plot a curve against.""" + ws = ctx.test_series(ctx.columns.wind_speed).to_numpy(dtype=float) + return ws if np.isfinite(ws).sum() >= _MIN_FINITE else None + + +def _ws_label(ctx: DiagnosticContext) -> str: + return f"{ctx.columns.wind_speed} @ {ctx.test_wtg}" + + +def _sig_label(ctx: DiagnosticContext, col: str) -> str: + return f"{col} @ {ctx.test_wtg}" + + +def _segmented_panel( + ax: plt.Axes, x: np.ndarray, y: np.ndarray, segments: Segments, *, xlabel: str, ylabel: str +) -> None: + """Scatter ``y`` vs ``x`` once per (label, mask, colour) segment.""" + for label, mask, color in segments: + ax.scatter(x[mask], y[mask], s=6, alpha=0.3, color=color, label=label) + ax.set_xlabel(xlabel) + ax.set_ylabel(ylabel) + apply_grid(ax) + + +def _ops_pair( + ctx: DiagnosticContext, + fig: plt.Figure, + grid: object, + *, + col: int, + x: np.ndarray, + x_label: str, + segments: Segments, + top: str | None, + bottom: str | None, +) -> None: + """Draw a shared-x column of two segmented panels (``top`` and ``bottom`` signals vs ``x``).""" + ax_top = fig.add_subplot(grid[0, col]) # type: ignore[index] + ax_bottom = fig.add_subplot(grid[1, col], sharex=ax_top) # type: ignore[index] + for ax, sig in ((ax_top, top), (ax_bottom, bottom)): + if sig is None or not ctx.has_column(sig): + ax.set_visible(False) + continue + y = ctx.test_series(sig).to_numpy(dtype=float) + _segmented_panel(ax, x, y, segments, xlabel=x_label, ylabel=_sig_label(ctx, sig)) + ax_top.tick_params(labelbottom=False) + ax_top.set_xlabel("") + + +def _ops_curve_figure( + ctx: DiagnosticContext, *, segments: Segments, title: str, filename: str, stage: str +) -> Path | None: + """Build the shared 2x3 operating-curve figure: power curve, plus pitch(top)/rpm(bottom) vs power and vs ws.""" + ws = _own_ws(ctx) + if ws is None or not ctx.has_column(ctx.columns.active_power): + return None + cols = ctx.columns + power = ctx.test_series(cols.active_power).to_numpy(dtype=float) + ws_label, power_label = _ws_label(ctx), _sig_label(ctx, cols.active_power) + + fig = plt.figure(figsize=(18, 9)) + grid = fig.add_gridspec(2, 3) + ax_pc = fig.add_subplot(grid[:, 0]) + _segmented_panel(ax_pc, ws, power, segments, xlabel=ws_label, ylabel=power_label) + ax_pc.set_title("power curve") + ax_pc.legend() + _ops_pair( + ctx, fig, grid, col=1, x=power, x_label=power_label, segments=segments, top=cols.pitch, bottom=cols.gen_rpm + ) + _ops_pair(ctx, fig, grid, col=2, x=ws, x_label=ws_label, segments=segments, top=cols.pitch, bottom=cols.gen_rpm) + + fig.suptitle(f"{ctx.test_wtg}: {title}") + path = ctx.stage_dir(stage) / filename + save_fig(fig, path) + return path + + +def plot_ops_curves(ctx: DiagnosticContext) -> Path | None: + """Draw the operating-curve figure coloured kept vs removed (the filter check).""" + used = np.asarray(ctx.used_ts, dtype=bool) + segments: Segments = [("kept", used, "C0"), ("removed", ~used, "C3")] + return _ops_curve_figure( + ctx, + segments=segments, + title="operating curves (kept vs removed by the row filter)", + filename="ops_curves.png", + stage=stages.FILTER, + ) + + +def plot_ops_curves_kept(ctx: DiagnosticContext) -> Path | None: + """Draw the operating-curve figure for the KEPT rows only (so removed points cannot mask them).""" + used = np.asarray(ctx.used_ts, dtype=bool) + segments: Segments = [("kept", used, "C0")] + return _ops_curve_figure( + ctx, + segments=segments, + title="operating curves (used rows only)", + filename="ops_curves_kept_only.png", + stage=stages.FILTER, + ) + + +def plot_ops_curves_excluded(ctx: DiagnosticContext) -> Path | None: + """Draw the operating-curve figure coloured kept vs caller-excluded (``ColumnSchema.exclude_row``). + + The same view as :func:`plot_ops_curves`, so a special operating mode reads as a band in + pitch/rpm. ``None`` when the method has no exclusion mask or excluded nothing. + """ + excluded = ctx.excluded_mask() + if excluded is None: + return None + segments: Segments = [("kept", ~excluded, "C0"), ("excluded", excluded, "C3")] + return _ops_curve_figure( + ctx, + segments=segments, + title=f"operating curves (kept vs excluded by the caller's flag; {excluded.sum()} rows excluded)", + filename="ops_curves_excluded.png", + stage=stages.FILTER, + ) + + +def plot_curves_by_upgrade(ctx: DiagnosticContext) -> Path | None: + """Draw the operating-curve figure coloured baseline vs upgraded, over the used rows.""" + used = np.asarray(ctx.used_ts, dtype=bool) + segments: Segments = [("baseline", used & ctx.baseline_ts, "C0"), ("upgraded", used & ctx.upgraded_ts, "C1")] + return _ops_curve_figure( + ctx, + segments=segments, + title="operating curves by upgrade state (used rows)", + filename="ops_curves_by_upgrade.png", + stage=stages.UPLIFT_INPUTS, + ) + + +def plot_reactive_vs_active(ctx: DiagnosticContext) -> Path | None: + """Reactive vs active power by upgrade state, one panel per turbine. None if no reactive tag.""" + if not ctx.has_column(ctx.columns.reactive_power): + return None + turbines = [ctx.test_wtg, *ctx.references()] + ncols = min(3, len(turbines)) + nrows = int(np.ceil(len(turbines) / ncols)) + fig, axes = plt.subplots(nrows, ncols, figsize=(5 * ncols, 4.5 * nrows), squeeze=False) + flat = axes.flatten() + for ax, turbine in zip(flat, turbines, strict=False): + active = ctx.turbine_series(turbine, ctx.columns.active_power).to_numpy(dtype=float) + reactive = ctx.turbine_series(turbine, ctx.columns.reactive_power).to_numpy(dtype=float) + for seg_label, seg, color in (("baseline", ctx.baseline_ts, "C0"), ("upgraded", ctx.upgraded_ts, "C1")): + ax.scatter(active[seg], reactive[seg], s=6, alpha=0.3, color=color, label=seg_label) + ax.set_title(f"{turbine}{' (test)' if turbine == ctx.test_wtg else ''}") + ax.set_xlabel(ctx.columns.active_power) + ax.set_ylabel(ctx.columns.reactive_power) # type: ignore[arg-type] + apply_grid(ax) + ax.legend() + for ax in flat[len(turbines) :]: + ax.set_visible(False) + fig.suptitle("reactive vs active power by upgrade state") + path = ctx.stage_dir(stages.INPUTS) / "reactive_vs_active.png" + save_fig(fig, path) + return path + + +def plot_power_factor(ctx: DiagnosticContext) -> Path | None: + """Active-power-weighted monthly power factor over time, per turbine. None if no reactive tag.""" + if not ctx.has_column(ctx.columns.reactive_power): + return None + fig, ax = plt.subplots(figsize=(12, 6)) + shade_segments(ax, ctx) + for turbine in [ctx.test_wtg, *ctx.references()]: + monthly = _monthly_power_factor(ctx, turbine) + label = f"{turbine}{' (test)' if turbine == ctx.test_wtg else ''}" + ax.plot(monthly.index.to_numpy(), monthly.to_numpy(), linewidth=1.0, marker=".", markersize=3, label=label) + ax.set_xlabel("date") + ax.set_ylabel("power factor (active-power-weighted monthly mean)") + ax.set_title("power factor over time — |P| / sqrt(P^2 + Q^2)") + apply_grid(ax) + ax.legend(ncol=2, fontsize="small") + path = ctx.stage_dir(stages.INPUTS) / "power_factor.png" + save_fig(fig, path) + return path + + +def _monthly_power_factor(ctx: DiagnosticContext, turbine: str) -> pd.Series: + """Monthly active-power-weighted mean power factor for one turbine.""" + active = ctx.turbine_series(turbine, ctx.columns.active_power) + reactive = ctx.turbine_series(turbine, ctx.columns.reactive_power) + apparent = np.sqrt(active**2 + reactive**2) + with np.errstate(divide="ignore", invalid="ignore"): + pf = (active.abs() / apparent).where(apparent > 0) + weight = active.abs() + num = (pf * weight).resample(_PF_BUCKET).sum(min_count=1) + den = weight.where(pf.notna()).resample(_PF_BUCKET).sum(min_count=1) + return num / den diff --git a/benchmarking/diagnostics/density.py b/benchmarking/diagnostics/density.py new file mode 100644 index 00000000..dbad13b9 --- /dev/null +++ b/benchmarking/diagnostics/density.py @@ -0,0 +1,102 @@ +"""Scatter plots coloured by data density. + +wind-up datasets are large, so plain scatter plots saturate and the eye cannot tell where the +bulk of the data sits versus a handful of outliers. Colouring each point by the local 2-D +histogram density fixes that. Small and degenerate inputs fall back to a flat colour rather than +raising, so a diagnostic plot never brings a run down. + +The density field is a fine 2-D histogram smoothed with a gaussian before it is sampled back onto +the points, giving a KDE-like gradient at O(n) cost (no per-point ``gaussian_kde``, which is +O(n^2) and prohibitively slow on wind-up-sized data). +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING, Any + +import numpy as np +from scipy.interpolate import interpn +from scipy.ndimage import gaussian_filter + +if TYPE_CHECKING: + import matplotlib.pyplot as plt + import numpy.typing as npt + +# splinef2d needs a few points along each axis; below this fall back to a flat colour. +_MIN_POINTS_FOR_DENSITY = 8 +_MIN_UNIQUE_PER_AXIS = 2 +_DEFAULT_BINS = 120 # fine histogram; the gaussian smoothing below keeps the gradient continuous +_DEFAULT_SMOOTH_SIGMA = 2.0 # gaussian smoothing (in bins) for a KDE-like gradient + + +def density_scatter( + x: npt.ArrayLike, + y: npt.ArrayLike, + *, + ax: plt.Axes, + bins: int = _DEFAULT_BINS, + smooth_sigma: float = _DEFAULT_SMOOTH_SIGMA, + sort: bool = True, + colorbar: bool = True, + **kwargs: Any, # noqa: ANN401 +) -> plt.Axes: + """Scatter ``y`` vs ``x`` on ``ax``, colouring points by smoothed 2-D histogram density. + + Non-finite pairs are dropped. With too few points to estimate a density the points are still + plotted (flat colour). The densest points are drawn last so they are visible on top. + + :param ax: the axes to draw on (required; callers manage the figure/layout) + :param bins: 2-D histogram bin count per axis + :param smooth_sigma: gaussian smoothing of the histogram, in bins (0 disables); a KDE-like gradient + :param sort: draw densest points last + :param colorbar: attach a "data density" colourbar to ``ax`` + """ + xv = np.asarray(x, dtype=float) + yv = np.asarray(y, dtype=float) + finite = np.isfinite(xv) & np.isfinite(yv) + xv, yv = xv[finite], yv[finite] + if len(xv) == 0: + return ax + + z = _density(xv, yv, bins=bins, smooth_sigma=smooth_sigma) + if sort: + order = z.argsort() + xv, yv, z = xv[order], yv[order], z[order] + + scatter = ax.scatter(xv, yv, c=z, **kwargs) + if colorbar: + cbar = ax.figure.colorbar(scatter, ax=ax) + # The density scale is arbitrary, so hide the numeric ticks. Use the colorbar API + # (set_ticks) rather than set_yticklabels([]), which trips Matplotlib's FixedFormatter + # warning — fatal under the tests' warnings-as-errors config. + cbar.set_ticks([]) + cbar.ax.set_ylabel("data density") + return ax + + +def _density( + x: npt.NDArray[np.float64], y: npt.NDArray[np.float64], *, bins: int, smooth_sigma: float = _DEFAULT_SMOOTH_SIGMA +) -> npt.NDArray[np.float64]: + """Per-point density via a smoothed 2-D histogram interpolated back onto the points (0 on failure).""" + if ( + len(x) < _MIN_POINTS_FOR_DENSITY + or len(np.unique(x)) < _MIN_UNIQUE_PER_AXIS + or len(np.unique(y)) < _MIN_UNIQUE_PER_AXIS + ): + return np.zeros(len(x)) + data, x_e, y_e = np.histogram2d(x, y, bins=bins, density=True) + if smooth_sigma > 0: + data = gaussian_filter(data, sigma=smooth_sigma) # KDE-like smooth gradient over the fine bins + # Linear (not splinef2d): points at the data edges fall just outside the bin-centre grid, and + # splinef2d refuses to extrapolate. Linear with a 0 fill is robust and visually equivalent. + z = interpn( + (0.5 * (x_e[1:] + x_e[:-1]), 0.5 * (y_e[1:] + y_e[:-1])), + data, + np.vstack([x, y]).T, + method="linear", + bounds_error=False, + fill_value=0.0, + ) + z = np.asarray(z, dtype=float) + z[~np.isfinite(z)] = 0.0 + return z diff --git a/benchmarking/diagnostics/histograms.py b/benchmarking/diagnostics/histograms.py new file mode 100644 index 00000000..bcf59328 --- /dev/null +++ b/benchmarking/diagnostics/histograms.py @@ -0,0 +1,132 @@ +"""Baseline-vs-upgraded condition histograms (feedback 2026-06-26, item 22). + +The prepost failure is an overlap/positivity problem: the baseline and upgraded segments do not +share a weather distribution. These overlaid histograms make that concrete — wind speed, +direction, temperature, turbulence, and the time-of-day / month structure — so a reviewer can +see exactly where the two segments differ. + +Hour-of-day and month are derived from the timestamp index and are **diagnostics only**: they +are deliberately not model features (the R-learner uses shuffled K-fold cross-fitting, which +assumes no timestamp features — design-note §3/§4). Weather conditions use reference-derived +(treatment-invariant) signals so the comparison is honest for the upgraded segment too. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np + +from benchmarking.diagnostics import stages +from benchmarking.diagnostics.context import ERA5_WD_COL +from benchmarking.diagnostics.style import apply_grid, save_fig + +if TYPE_CHECKING: + from pathlib import Path + + from benchmarking.diagnostics.context import DiagnosticContext + +_MONTH_LABELS = ["Jan", "Feb", "Mar", "Apr", "May", "Jun", "Jul", "Aug", "Sep", "Oct", "Nov", "Dec"] + +# Real Open-Meteo wind-direction column, preferred over the neutral ``era5_wd`` alias for labels. +_ERA5_RAW_WD = "wind_direction_100m" + + +def _era5_wd_column(ctx: DiagnosticContext) -> str | None: + """Return the ERA5 wind-direction column to plot — the real Open-Meteo name if present, else the alias.""" + if ctx.era5_df is None: + return None + for col in (_ERA5_RAW_WD, ERA5_WD_COL): + if col in ctx.era5_df.columns: + return col + return None + + +def _conditions(ctx: DiagnosticContext) -> list[tuple[str, np.ndarray, np.ndarray | None]]: + """Return the (label, per-timestamp values, bins) conditions available for this run. + + ``bins`` is an explicit edge array for the discrete time features (hour, month) and ``None`` + for continuous ones (their edges are derived robustly at plot time so low-wind TI outliers do + not flatten the histogram). + """ + cols = ctx.columns + index = ctx.index + items: list[tuple[str, np.ndarray, np.ndarray | None]] = [ + (f"{cols.wind_speed} (ref mean) [m/s]", ctx.reference_mean(cols.wind_speed).to_numpy(dtype=float), None), + ] + + wd_col = _era5_wd_column(ctx) + if wd_col is not None and ctx.era5_df is not None: + items.append((f"{wd_col} [deg]", ctx.era5_df[wd_col].reindex(index).to_numpy(dtype=float), None)) + elif ctx.has_column(cols.nacelle_position): + label = f"{cols.nacelle_position} (ref mean) [deg]" + items.append((label, ctx.reference_mean(cols.nacelle_position).to_numpy(dtype=float), None)) + + if ctx.has_column(cols.wind_speed_sd): + ws = ctx.reference_mean(cols.wind_speed).to_numpy(dtype=float) + ws_sd = ctx.reference_mean(cols.wind_speed_sd).to_numpy(dtype=float) + with np.errstate(divide="ignore", invalid="ignore"): + ti = np.where(ws > 0, ws_sd / ws, np.nan) + items.append((f"TI = {cols.wind_speed_sd}/{cols.wind_speed} (ref mean, dimensionless)", ti, None)) + + if ctx.has_column(cols.ambient_temp): + items.append( + (f"{cols.ambient_temp} (ref mean)", ctx.reference_mean(cols.ambient_temp).to_numpy(dtype=float), None) + ) + + items.append(("hour of day", index.hour.to_numpy(dtype=float), np.arange(-0.5, 24.5, 1.0))) + items.append(("month", index.month.to_numpy(dtype=float), np.arange(0.5, 13.5, 1.0))) + return items + + +def _robust_bins(values: np.ndarray, *, bins: int = 30) -> np.ndarray | int: + """Bin edges over the 1st-99th percentile of the finite values, so outliers don't dominate.""" + finite = values[np.isfinite(values)] + if finite.size == 0: + return bins + lo, hi = np.percentile(finite, [1, 99]) + if hi <= lo: + return bins + return np.linspace(lo, hi, bins + 1) + + +def plot_condition_histograms(ctx: DiagnosticContext) -> Path: + """Overlaid baseline-vs-upgraded histograms for each available condition.""" + conditions = _conditions(ctx) + baseline = ctx.baseline_ts + upgraded = ctx.upgraded_ts + ncols = 3 + nrows = int(np.ceil(len(conditions) / ncols)) + fig, axes = plt.subplots(nrows, ncols, figsize=(5 * ncols, 4 * nrows), squeeze=False) + flat = axes.flatten() + for ax, (label, values, explicit_bins) in zip(flat, conditions, strict=False): + bins = explicit_bins if explicit_bins is not None else _robust_bins(values) + for seg_label, seg, color in (("baseline", baseline, "C0"), ("upgraded", upgraded, "C1")): + seg_vals = values[seg & np.isfinite(values)] + if seg_vals.size: + # filled with alpha so a coincident distribution (e.g. flat hour-of-day) shows both. + ax.hist( + seg_vals, bins=bins, density=True, histtype="stepfilled", alpha=0.45, color=color, label=seg_label + ) + ax.set_xlabel(label) + ax.set_ylabel("density") + _label_time_axis(ax, label) + apply_grid(ax) + if ax.get_legend_handles_labels()[0]: + ax.legend() + for ax in flat[len(conditions) :]: + ax.set_visible(False) + fig.suptitle(f"{ctx.test_wtg}: condition distributions on the USED (post-filter) data, baseline vs upgraded") + path = ctx.stage_dir(stages.UPLIFT_INPUTS) / "condition_histograms.png" + save_fig(fig, path) + return path + + +def _label_time_axis(ax: plt.Axes, label: str) -> None: + """Nicely tick the discrete hour-of-day and month axes.""" + if label == "hour of day": + ax.set_xticks(range(0, 24, 3)) + elif label == "month": + ax.set_xticks(range(1, 13)) + ax.set_xticklabels(_MONTH_LABELS, rotation=45, fontsize="small") diff --git a/benchmarking/diagnostics/northing.py b/benchmarking/diagnostics/northing.py new file mode 100644 index 00000000..d93f6be6 --- /dev/null +++ b/benchmarking/diagnostics/northing.py @@ -0,0 +1,75 @@ +"""Northing-error timeline (feedback 2026-06-26, item 15). + +The R-learner gets reference nacelle positions as raw features with **no** northing correction; +an offset or jumps in a turbine's yaw zero distorts the direction signal the model sees. This +plots, per turbine, the **monthly circular mean** of (nacelle position - ERA5 wind direction) over +time, so a drift or step in the offset stands out. Only rows where the turbine is generating +(≥ 5% of its rated power) are used, because a parked turbine often points away from the wind. + +Requires a nacelle-position column and aligned ERA5 direction; returns ``None`` otherwise. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +from benchmarking.diagnostics import stages +from benchmarking.diagnostics.context import ERA5_WD_COL +from benchmarking.diagnostics.style import apply_grid, save_fig +from benchmarking.diagnostics.timeaxis import shade_segments + +if TYPE_CHECKING: + from pathlib import Path + + from benchmarking.diagnostics.context import DiagnosticContext + +_GENERATING_FRAC = 0.05 # keep rows above 5% of (proxy) rated power +_RATED_PERCENTILE = 99 # robust proxy for rated power + + +def _wrap180(deg: pd.Series) -> pd.Series: + """Wrap an angle (degrees) to the [-180, 180) range.""" + return (deg + 180.0) % 360.0 - 180.0 + + +def _monthly_circular_mean(error_deg: pd.Series) -> pd.Series: + """Per-month circular mean (degrees) of an angle series, NaN-skipping.""" + rad = np.deg2rad(error_deg) + sin = np.sin(rad).resample("MS").mean() + cos = np.cos(rad).resample("MS").mean() + return pd.Series(np.rad2deg(np.arctan2(sin, cos)), index=sin.index) + + +def plot_northing_error(ctx: DiagnosticContext) -> Path | None: + """Per-turbine monthly circular-mean of (nacelle position - ERA5 direction) over time.""" + if ( + not ctx.has_column(ctx.columns.nacelle_position) + or ctx.era5_df is None + or ERA5_WD_COL not in ctx.era5_df.columns + ): + return None + era5_wd = ctx.era5_df[ERA5_WD_COL].reindex(ctx.index) + fig, ax = plt.subplots(figsize=(12, 6)) + shade_segments(ax, ctx) + for turbine in [ctx.test_wtg, *ctx.references()]: + nacelle = ctx.turbine_series(turbine, ctx.columns.nacelle_position) + power = ctx.turbine_series(turbine, ctx.columns.active_power) + rated = np.nanpercentile(power.to_numpy(dtype=float), _RATED_PERCENTILE) if power.notna().any() else np.nan + generating = power >= _GENERATING_FRAC * rated if np.isfinite(rated) else power.notna() + error = _wrap180(nacelle - era5_wd).where(generating) + monthly = _monthly_circular_mean(error) + label = f"{turbine}{' (test)' if turbine == ctx.test_wtg else ''}" + ax.plot(monthly.index.to_numpy(), monthly.to_numpy(), linewidth=1.0, marker=".", markersize=3, label=label) + ax.axhline(0.0, color="k", linewidth=1) + ax.set_xlabel("date") + ax.set_ylabel(f"{ctx.columns.nacelle_position} - {ERA5_WD_COL} [deg] (monthly circular mean)") + ax.set_title(f"{ctx.test_wtg}: northing error over time (generating rows only, no corrections)") + apply_grid(ax) + ax.legend(ncol=2, fontsize="small") + path = ctx.stage_dir(stages.INPUTS) / "northing_error.png" + save_fig(fig, path) + return path diff --git a/benchmarking/diagnostics/stages.py b/benchmarking/diagnostics/stages.py new file mode 100644 index 00000000..72ee1184 --- /dev/null +++ b/benchmarking/diagnostics/stages.py @@ -0,0 +1,16 @@ +"""Analysis-stage folder names for grouping the per-run diagnostic plots. + +Plots are written into numbered stage subfolders of ``/plots`` so a reviewer can tell at a +glance which step of the pipeline a plot describes (feedback 2026-06-26): raw inputs, filtering, +feature engineering, the data fed to the uplift model, the modelling itself, and the results. +""" + +from __future__ import annotations + +INPUTS = "1_inputs" +FILTER = "2_filter" +FEATURE_ENG = "3_feature_eng" +UPLIFT_INPUTS = "4_uplift_inputs" +UPLIFT_MODELLING = "5_uplift_modelling" +UPLIFT_RESULTS = "6_uplift_results" +CONDITIONAL_UPLIFT = "7_conditional_uplift" diff --git a/benchmarking/diagnostics/style.py b/benchmarking/diagnostics/style.py new file mode 100644 index 00000000..30612286 --- /dev/null +++ b/benchmarking/diagnostics/style.py @@ -0,0 +1,48 @@ +"""Shared plotting conventions for the benchmarking diagnostics. + +One place to enforce the project-wide rules (feedback 2026-06-26): a grid on every axes unless +there is a good reason not to, and a single ``save_fig`` that tight-lays-out, writes at a +consistent DPI and closes the figure. +""" + +from __future__ import annotations + +import contextlib +import os +import sys +from typing import TYPE_CHECKING + +import matplotlib as mpl + +# Headless by default: these run in studies/CI with no display. Mirror wind_up/__init__.py — respect +# an explicit MPLBACKEND (checked by key presence), leave the backend alone if pyplot is already +# imported (e.g. an interactive notebook), and never let a late use() failure break import. +if "MPLBACKEND" not in os.environ and "matplotlib.pyplot" not in sys.modules: + with contextlib.suppress(ImportError): + mpl.use("Agg") + +if TYPE_CHECKING: + from pathlib import Path + + import matplotlib.pyplot as plt + +_DPI = 150 +_GRID_ALPHA = 0.3 + + +def apply_grid(ax: plt.Axes) -> None: + """Turn on a light grid (the project default for every axes).""" + ax.grid(visible=True, alpha=_GRID_ALPHA) + + +def save_fig(fig: plt.Figure, path: Path) -> None: + """Write ``fig`` to ``path`` at the standard DPI (tight bbox) and close it. + + Uses ``bbox_inches="tight"`` rather than ``tight_layout`` so figures with colorbars/imshow + (which ``tight_layout`` warns about — and tests treat warnings as errors) lay out cleanly. + """ + path.parent.mkdir(parents=True, exist_ok=True) + fig.savefig(path, dpi=_DPI, bbox_inches="tight") + import matplotlib.pyplot as plt # noqa: PLC0415 + + plt.close(fig) diff --git a/benchmarking/diagnostics/timeaxis.py b/benchmarking/diagnostics/timeaxis.py new file mode 100644 index 00000000..bf229614 --- /dev/null +++ b/benchmarking/diagnostics/timeaxis.py @@ -0,0 +1,34 @@ +"""Shared helpers for time-axis diagnostic plots: the upgrade boundary and segment shading.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import numpy as np + +if TYPE_CHECKING: + import matplotlib.pyplot as plt + import pandas as pd + + from benchmarking.diagnostics.context import DiagnosticContext + +BASELINE_COLOR = "#59a89c" +UPGRADED_COLOR = "#e08214" + + +def upgrade_start(ctx: DiagnosticContext) -> pd.Timestamp | None: + """Return the first upgraded timestamp (prepost changeover / toggle origin), or None.""" + upgraded = ctx.upgraded_ts + return ctx.index[int(np.argmax(upgraded))] if upgraded.any() else None + + +def shade_segments(ax: plt.Axes, ctx: DiagnosticContext) -> None: + """Shade the baseline and upgraded spans on a time axis and mark the upgrade start.""" + start = upgrade_start(ctx) + first, last = ctx.index.min(), ctx.index.max() + if start is not None: + ax.axvspan(first, start, color=BASELINE_COLOR, alpha=0.12) + ax.axvspan(start, last, color=UPGRADED_COLOR, alpha=0.12) + ax.axvline(start, color="k", linestyle="--", linewidth=1.2) + else: + ax.axvspan(first, last, color=BASELINE_COLOR, alpha=0.12) diff --git a/benchmarking/harness/__init__.py b/benchmarking/harness/__init__.py new file mode 100644 index 00000000..dda81c97 --- /dev/null +++ b/benchmarking/harness/__init__.py @@ -0,0 +1,69 @@ +"""P50 evaluation harness (v1 benchmarking, WS1). + +Score an uplift method's P50 estimate against the known injected truth from the synthetic +generator, reporting accuracy (bias) and precision (spread) as a function of campaign length. +""" + +from __future__ import annotations + +from benchmarking.harness.calibration import ( + TARGET_COVERAGE_1SIGMA, + CalibrationSummary, + calibration_summary, + coverage_standard_error, + summarize_calibration, +) +from benchmarking.harness.campaign import ( + CampaignWindow, + campaign_windows, + treated_activity_mask, + window_row_mask, +) +from benchmarking.harness.conditions import ( + CONDITION_BINS, + CONDITIONS, + TI_BINS, + WS_BINS, + condition_bins, + energy_ratio_by_bin, +) +from benchmarking.harness.leaderboard import conditional_leaderboard, leaderboard +from benchmarking.harness.method import Method, MethodInput, MethodOutput +from benchmarking.harness.metrics import ErrorSummary, summarize_errors +from benchmarking.harness.plots import plot_campaign_curves, plot_conditional_uplift +from benchmarking.harness.replicates import Replicate, StudyConfig, build_replicates, iter_replicates +from benchmarking.harness.scoring import score_one, score_study, truth_mask + +__all__ = [ + "CONDITIONS", + "CONDITION_BINS", + "TARGET_COVERAGE_1SIGMA", + "TI_BINS", + "WS_BINS", + "CalibrationSummary", + "CampaignWindow", + "ErrorSummary", + "Method", + "MethodInput", + "MethodOutput", + "Replicate", + "StudyConfig", + "build_replicates", + "calibration_summary", + "campaign_windows", + "condition_bins", + "conditional_leaderboard", + "coverage_standard_error", + "energy_ratio_by_bin", + "iter_replicates", + "leaderboard", + "plot_campaign_curves", + "plot_conditional_uplift", + "score_one", + "score_study", + "summarize_calibration", + "summarize_errors", + "treated_activity_mask", + "truth_mask", + "window_row_mask", +] diff --git a/benchmarking/harness/calibration.py b/benchmarking/harness/calibration.py new file mode 100644 index 00000000..4dda39db --- /dev/null +++ b/benchmarking/harness/calibration.py @@ -0,0 +1,142 @@ +"""Score an uncertainty: whether a reported sigma matches the error it claims to describe. + +The metrics in :mod:`benchmarking.harness.metrics` ask how close an estimate lands to truth; these +ask whether the method was right about how close it would land. The statistic is the standardised +error ``z = signed_error / sigma``, which a calibrated 1-sigma makes a standard normal: ~68.3% +within 1 sigma, ``std(z) == 1``. + +Sigma is scored against the deviation from **ground truth**, which includes the method's bias. A +sigma covering only sampling variance will under-cover wherever the method is biased — that is a +finding about the uncertainty, not an unfairness in the metric. + +Coverage, ``z_spread`` and ``z_robust`` are reported together because they fail differently, and +disagreement localises the problem (``z_spread >> z_robust`` means the tails, not the typical case). +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd + +if TYPE_CHECKING: + from collections.abc import Sequence + + import numpy.typing as npt + +# P(|Z| <= 1) for a standard normal: the fraction of estimates a calibrated 1-sigma should cover. +TARGET_COVERAGE_1SIGMA = 0.6827 +# median(|x - median(x)|) * this == std(x) for a normal x; scales the MAD onto a sigma. +_MAD_TO_SIGMA = 1.4826 + + +@dataclass(frozen=True) +class CalibrationSummary: + """How well a reported sigma described the errors it was reported alongside. + + :param coverage_1sigma: fraction of cases with ``|z| <= 1``; target :data:`TARGET_COVERAGE_1SIGMA` + :param z_spread: population std of ``z``; target 1.0. Above 1 means sigma was too small. + :param z_robust: ``1.4826 * MAD(z)``; target 1.0. The outlier-resistant counterpart of ``z_spread``. + :param mean_sigma: mean reported sigma — the interval width, so inflating sigma to win coverage + shows up here + :param rms_error: RMS of the signed errors; the scale ``mean_sigma`` should be near + :param n: usable cases (finite error, finite and strictly positive sigma) + :param n_unusable: cases with a finite error but an unusable sigma. Counted, not silently + dropped: a collapsed sigma is an uncertainty failure. + """ + + coverage_1sigma: float + z_spread: float + z_robust: float + mean_sigma: float + rms_error: float + n: int + n_unusable: int + + +def calibration_summary(signed_errors: npt.ArrayLike, sigmas: npt.ArrayLike) -> CalibrationSummary: + """Summarise how well ``sigmas`` describe ``signed_errors``. + + A non-finite error is ignored (nothing to calibrate against). A finite error with an unusable + sigma is excluded but counted in ``n_unusable``. Empty or fully unusable input gives a NaN + summary with ``n == 0``. + """ + errors = np.asarray(signed_errors, dtype=float).ravel() + sigma = np.asarray(sigmas, dtype=float).ravel() + if errors.shape != sigma.shape: + msg = f"signed_errors and sigmas must be the same length; got {errors.shape} and {sigma.shape}" + raise ValueError(msg) + + scoreable = np.isfinite(errors) + usable = scoreable & np.isfinite(sigma) & (sigma > 0) + n_unusable = int((scoreable & ~usable).sum()) + if not usable.any(): + nan = float("nan") + return CalibrationSummary( + coverage_1sigma=nan, z_spread=nan, z_robust=nan, mean_sigma=nan, rms_error=nan, n=0, n_unusable=n_unusable + ) + + err = errors[usable] + sig = sigma[usable] + z = err / sig + return CalibrationSummary( + coverage_1sigma=float(np.mean(np.abs(z) <= 1.0)), + z_spread=float(z.std(ddof=0)), + z_robust=float(_MAD_TO_SIGMA * np.median(np.abs(z - np.median(z)))), + mean_sigma=float(sig.mean()), + rms_error=float(np.sqrt(np.mean(err**2))), + n=int(err.size), + n_unusable=n_unusable, + ) + + +def summarize_calibration( + results_df: pd.DataFrame, + *, + group_keys: Sequence[str], + error_col: str = "signed_error", + sigma_col: str = "sigma", +) -> pd.DataFrame: + """Reduce tidy results to one calibration row per ``group_keys`` group. + + :param results_df: tidy scoring results carrying ``error_col`` and ``sigma_col`` + :param group_keys: the columns to group by (e.g. ``["block_hours", "campaign_weeks"]``) + """ + keys = list(group_keys) + missing = [c for c in [*keys, error_col, sigma_col] if c not in results_df.columns] + if missing: + msg = f"results_df is missing column(s) {missing}" + raise ValueError(msg) + + records = [] + for values, group in results_df.groupby(keys, sort=True, dropna=False): + summary = calibration_summary(group[error_col].to_numpy(), group[sigma_col].to_numpy()) + key_values = values if isinstance(values, tuple) else (values,) + records.append( + { + **dict(zip(keys, key_values, strict=True)), + "coverage_1sigma": summary.coverage_1sigma, + "z_spread": summary.z_spread, + "z_robust": summary.z_robust, + "mean_sigma": summary.mean_sigma, + "rms_error": summary.rms_error, + "n": summary.n, + "n_unusable": summary.n_unusable, + } + ) + columns = [*keys, "coverage_1sigma", "z_spread", "z_robust", "mean_sigma", "rms_error", "n", "n_unusable"] + return pd.DataFrame(records, columns=columns) + + +def coverage_standard_error(n: int, *, coverage: float = TARGET_COVERAGE_1SIGMA) -> float: + """Binomial standard error of a coverage estimate from ``n`` **independent** cases. + + Supplying ``n`` is the caller's job, and a sweep's row count is not it: profiles sharing campaign + windows, prefix-nested campaign lengths, and overlapping long campaigns all multiply rows without + adding independent evidence. Passing a row count here understates the error badly. + """ + if n <= 0: + return float("nan") + return float(np.sqrt(coverage * (1.0 - coverage) / n)) diff --git a/benchmarking/harness/campaign.py b/benchmarking/harness/campaign.py new file mode 100644 index 00000000..119a5cc5 --- /dev/null +++ b/benchmarking/harness/campaign.py @@ -0,0 +1,154 @@ +"""Campaign-length windows and the two record selections each window implies. + +The short-campaign sweep varies the treatment-activity duration (post window for prepost, +toggling duration for toggle) over a grid — e.g. ``{3, 6, 9, 12}`` months, or ``{1, 2, 4, 8}`` +weeks where a month is too coarse a step — while holding a fixed baseline before the treatment +start. Every length shares the one ``(treatment_start, baseline_start)`` and differs only in +``activity_end``, so shorter windows are strict leading prefixes of longer ones — the +campaign-length curve isolates the effect of post-duration alone. + +A grid is in **either** months or weeks, never both; the unit travels with the window +(:attr:`CampaignWindow.unit`) so results are reported under the right column name. + +From a window two distinct selections are derived, deliberately kept apart: + +- :func:`window_row_mask` — the method-facing rows (all turbines, baseline + activity). +- :func:`treated_activity_mask` — the test turbine's treated rows within the activity window, + for the ground-truth comparison. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING, Literal + +import numpy as np +import pandas as pd + +from benchmarking.synthetic import treated_mask + +if TYPE_CHECKING: + import numpy.typing as npt + + from benchmarking.synthetic import ToggleSchedule + +CampaignUnit = Literal["months", "weeks"] + + +@dataclass(frozen=True) +class CampaignWindow: + """One campaign length: a fixed baseline then ``length`` ``unit`` of treatment activity. + + The activity length is expressed in either months or weeks. Months is the original grid (the + overnight studies and their frozen baselines); weeks exists for short campaigns, where a month + is too coarse a step — a real toggle campaign runs for a handful of weeks. + + :param length: the campaign-length grid value + :param unit: the grid's unit, ``"months"`` or ``"weeks"`` + """ + + length: int + unit: CampaignUnit + baseline_start: pd.Timestamp + treatment_start: pd.Timestamp + activity_end: pd.Timestamp + + @property + def months(self) -> int: + """The activity length in months. Raises for a weeks window. + + Months-only consumers read this. It raises rather than returning ``length`` so a weeks window + reaching a months-only caller fails loudly instead of silently reporting a week count as a + month count. + """ + if self.unit != "months": + msg = f"CampaignWindow.months is only defined for a months grid, but this window has unit={self.unit!r}" + raise ValueError(msg) + return self.length + + @property + def length_col(self) -> str: + """The result column this window's length is reported under (``campaign_months``/``_weeks``).""" + return f"campaign_{self.unit}" + + +def resolve_campaign_grid( + *, campaign_months: list[int] | None, campaign_weeks: list[int] | None +) -> tuple[list[int], CampaignUnit]: + """Return the ``(lengths, unit)`` of whichever grid is set. Raises unless exactly one is. + + Shared by :func:`campaign_windows` and ``StudyConfig`` so the two agree on the rule. + + Keyword-only deliberately: the two parameters have the same type, so a positional call would give + the reader nothing to check against, and transposing them would silently mislabel weeks as months + rather than fail. + """ + if (campaign_months is None) == (campaign_weeks is None): + msg = ( + "exactly one of campaign_months / campaign_weeks must be set, got " + f"campaign_months={campaign_months!r}, campaign_weeks={campaign_weeks!r}" + ) + raise ValueError(msg) + if campaign_months is not None: + return campaign_months, "months" + assert campaign_weeks is not None # noqa: S101 - narrowed by the check above + return campaign_weeks, "weeks" + + +def campaign_windows( + treatment_start: pd.Timestamp, + *, + min_pre_months: int, + campaign_months: list[int] | None = None, + campaign_weeks: list[int] | None = None, + data_start: pd.Timestamp | None = None, + data_end: pd.Timestamp | None = None, +) -> list[CampaignWindow]: + """Build one :class:`CampaignWindow` per campaign length. + + Exactly one of ``campaign_months`` / ``campaign_weeks`` gives the grid. + ``baseline_start = treatment_start - min_pre_months`` is fixed across lengths (and is always in + months, independent of the grid's unit); ``activity_end = treatment_start + length`` grows with + the length. When ``data_start`` / ``data_end`` are given, lengths whose window would fall outside + the available data are dropped (infeasible ``(replicate, length)`` combinations). + """ + lengths, unit = resolve_campaign_grid(campaign_months=campaign_months, campaign_weeks=campaign_weeks) + baseline_start = treatment_start - pd.DateOffset(months=min_pre_months) + windows = [] + for length in lengths: + activity_end = treatment_start + pd.DateOffset(**{unit: length}) + if data_start is not None and baseline_start < data_start: + continue + if data_end is not None and activity_end > data_end: + continue + windows.append( + CampaignWindow( + length=length, + unit=unit, + baseline_start=baseline_start, + treatment_start=treatment_start, + activity_end=activity_end, + ) + ) + return windows + + +def window_row_mask(index: pd.DatetimeIndex, window: CampaignWindow) -> npt.NDArray[np.bool_]: + """Boolean mask of the rows a method sees: ``[baseline_start, activity_end)``.""" + return np.asarray((index >= window.baseline_start) & (index < window.activity_end)) + + +def treated_activity_mask( + index: pd.DatetimeIndex, + upgrade_timing: pd.Timestamp | ToggleSchedule, + *, + window: CampaignWindow, +) -> npt.NDArray[np.bool_]: + """Boolean mask of the test turbine's treated rows within ``[treatment_start, activity_end)``. + + For prepost these are the post records; for toggle the on-records. The baseline is never + treated. ``index`` must be the test turbine's rows in time order (matches ``true_uplift``). + """ + treated = treated_mask(index, upgrade_timing) + in_activity = np.asarray((index >= window.treatment_start) & (index < window.activity_end)) + return treated & in_activity diff --git a/benchmarking/harness/conditions.py b/benchmarking/harness/conditions.py new file mode 100644 index 00000000..98f5cb03 --- /dev/null +++ b/benchmarking/harness/conditions.py @@ -0,0 +1,105 @@ +"""Shared condition bins and the binned energy-ratio reducer. + +One source of truth for the wind-speed / TI bin edges, imported by both the method (to bin its +counterfactual ledger) and the harness truth path, so the two bin identically. The reducer is the +same energy ratio as the overall number, taken within each bin: ``Σactual / Σcounterfactual - 1``. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import numpy as np +import numpy.typing as npt +import pandas as pd + +if TYPE_CHECKING: + from collections.abc import Iterable + +WS_BINS: list[float] = [float(x) for x in np.arange(0.0, 28.0, 2.0)] # 0,2,…,26 +TI_BINS: list[float] = [round(float(x), 2) for x in np.arange(0.0, 0.55, 0.05)] # 0,0.05,…,0.50 +CONDITIONS: tuple[str, ...] = ("ws", "ti", "power") +CONDITION_BINS: dict[str, list[float]] = {"ws": WS_BINS, "ti": TI_BINS} + +# ``power`` edges are a fraction of rated (so they scale with the turbine), unlike the fixed ws/ti +# edges, hence they live in ``condition_bins()`` not ``CONDITION_BINS``. The outer edges sit just +# below 0 and just above rated so the six bins center on 0,0.2,…,1.0 of rated and ``pd.cut`` keeps +# baseline power values that fall slightly negative (cut-in noise) or slightly over rated +# (counterfactual prediction noise) rather than dropping them to NaN. +POWER_FRACTION_EDGES: list[float] = [-0.1, 0.1, 0.3, 0.5, 0.7, 0.9, 1.1] + + +def validate_conditions(conditions: Iterable[str], *, supported: Iterable[str], method_name: str) -> None: + """Raise if any requested condition is unknown, or known but unsupported by this method. + + The two failures are distinguished deliberately: an unknown axis is a typo, while a known axis a + method cannot offer is a design limit of that method (e.g. ``toggle_specialist`` reports only + ``power`` — binning it by the test turbine's wind speed would break the property that it never + conditions on post-treatment signals). Conflating them would send the reader hunting for a + spelling mistake that is not there. + + An empty ``conditions`` is valid: it means "report no per-condition breakdown". + """ + supported = tuple(supported) + for name in conditions: + if name not in CONDITIONS: + msg = f"unknown condition {name!r}; expected one of {CONDITIONS}" + raise ValueError(msg) + if name not in supported: + msg = f"{method_name} does not support the {name!r} condition; it supports {supported}" + raise ValueError(msg) + + +def condition_bins(name: str, *, rated_power_kw: float | None = None) -> list[float]: + """Bin edges for a condition axis. ws/ti are fixed; power scales with ``rated_power_kw``. + + ws/ti accept ``rated_power_kw`` but ignore it (they are treatment-invariant fixed edges). power + requires it — its edges are ``POWER_FRACTION_EDGES`` times the rating (in kW). Raises for an + unknown condition so a mistyped axis fails loudly. + """ + if name in CONDITION_BINS: + return CONDITION_BINS[name] + if name == "power": + if rated_power_kw is None: + msg = "rated_power_kw is required for the 'power' condition (edges scale with rating)" + raise ValueError(msg) + return [round(f * float(rated_power_kw), 6) for f in POWER_FRACTION_EDGES] + msg = f"unknown condition {name!r}; expected one of {CONDITIONS}" + raise ValueError(msg) + + +def energy_ratio_by_bin( + condition_values: npt.ArrayLike, + actual: npt.ArrayLike, + counterfactual: npt.ArrayLike, + *, + bins: list[float], +) -> pd.DataFrame: + """Energy-ratio uplift within bins of a condition signal (NaN-safe; every bin represented).""" + cond = np.asarray(condition_values, dtype=float) + act = np.asarray(actual, dtype=float) + cf = np.asarray(counterfactual, dtype=float) + finite = np.isfinite(cond) & np.isfinite(act) & np.isfinite(cf) + frame = pd.DataFrame( + { + "condition_bin": pd.cut(cond[finite], bins=bins).astype(str), + "actual": act[finite], + "counterfactual": cf[finite], + } + ) + all_bins = pd.cut([], bins=bins).categories.astype(str) + grouped = frame.groupby("condition_bin", observed=False) + table = grouped[["actual", "counterfactual"]].sum() + table["n_records"] = grouped.size() + table = table.reindex(all_bins) + table["n_records"] = table["n_records"].fillna(0).astype(int) + # empty bins get a 0 energy sum (they contribute nothing to a downstream aggregation) + table[["actual", "counterfactual"]] = table[["actual", "counterfactual"]].fillna(0.0) + denom = table["counterfactual"].to_numpy() + table["p50_uplift"] = ( + np.divide(table["actual"].to_numpy(), denom, out=np.full(len(table), np.nan), where=denom != 0) - 1.0 + ) + table = table.rename(columns={"actual": "sum_actual", "counterfactual": "sum_counterfactual"}) + return table.reset_index(names="condition_bin")[ + ["condition_bin", "p50_uplift", "n_records", "sum_actual", "sum_counterfactual"] + ] diff --git a/benchmarking/harness/example_hot_study.py b/benchmarking/harness/example_hot_study.py new file mode 100644 index 00000000..3f1a0fd3 --- /dev/null +++ b/benchmarking/harness/example_hot_study.py @@ -0,0 +1,259 @@ +"""Driver: run the P50 evaluation harness end-to-end on real Hill of Towie data. + +A runnable, inspectable companion to ``tests/.../test_hot_end_to_end.py`` (which saves to a +pytest ``tmp_path`` and then throws it away). This wires the open Hill of Towie SCADA through +the full harness — load -> inject a constant-Cp upgrade -> replicate ensemble -> campaign +sweep -> score -> leaderboard -> plot — and saves the leaderboard CSV, the tidy per-replicate +results and the campaign-length curve PNG to a directory you can open. + +The "methods" scored here are illustrative stand-ins (no real estimator ships in this phase): + +- ``oracle`` returns the injected truth, so its error is ~0 at every campaign length; +- ``biased`` adds a fixed offset, so its error sits at that offset; +- ``realistic`` adds noise that shrinks with campaign length, so its spread band narrows as + the campaign grows — the precision-vs-data-volume effect the campaign sweep exists to show. + +Run it:: + + uv run python -m benchmarking.harness.example_hot_study + +First run downloads and caches the Hill of Towie v2 year zips from Zenodo (a few GB; a +12-month baseline plus a 12-month campaign spans 2016..2018). Override the window, output and +cache directories via the ``main`` arguments or the ``WIND_UP_BENCHMARKING_*`` env vars. +""" + +from __future__ import annotations + +import logging +import os +from pathlib import Path + +import numpy as np +import pandas as pd + +from benchmarking.harness import ( + Method, + MethodInput, + MethodOutput, + StudyConfig, + leaderboard, + plot_campaign_curves, + score_study, +) +from benchmarking.synthetic import HOT_COLUMNS, ConstantCpChange, treated_mask +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +logger = logging.getLogger(__name__) + +# A stable, no-upgrade Hill of Towie window wide enough for a 12-month baseline plus a +# 12-month campaign with the changeover at the 2017 new year. All of 2016-2020 was previously +# confirmed stable for these turbines (well before the real T13 AeroUp, installed Sep 2021). +DEFAULT_START_DT = pd.Timestamp("2016-01-01", tz="UTC") +DEFAULT_END_DT_EXCL = pd.Timestamp("2018-02-01", tz="UTC") +# The stable south-west turbines used by the end-to-end test (T01, T03, T04, T07). +DEFAULT_WTG_NUMBERS = [1, 3, 4, 7] +DEFAULT_TURBINE_SUBSET = ["T01", "T03", "T04", "T07"] + + +def default_output_root() -> Path: + """Return the directory the harness example writes its outputs under. + + Overridable via the ``WIND_UP_BENCHMARKING_OUTPUT_DIR`` environment variable; defaults to + ``~/temp/wind-up-benchmarking/harness``. + """ + root = Path(os.getenv("WIND_UP_BENCHMARKING_OUTPUT_DIR", Path.home() / "temp" / "wind-up-benchmarking")) + return root / "harness" + + +def _oracle_overall_uplift(mi: MethodInput, original_df: pd.DataFrame) -> float: + """Energy-ratio uplift over the treated test-turbine rows in ``mi``'s window. + + Compares the (upgraded) synthetic power against the original no-upgrade power over the + exact same treated records, so it recovers the injected truth. The illustrative methods + below wrap this to drive bias and spread. + """ + syn = mi.scada_df + test_rows = syn[syn[HOT_COLUMNS.turbine] == mi.test_wtg] + treated = treated_mask(test_rows.index, mi.upgrade_timing) + treated_rows = test_rows[treated] + + syn_power = treated_rows[HOT_COLUMNS.active_power].to_numpy(dtype=float) + orig_test = original_df[original_df[HOT_COLUMNS.turbine] == mi.test_wtg] + orig_power = orig_test.loc[treated_rows.index, HOT_COLUMNS.active_power].to_numpy(dtype=float) + + finite = np.isfinite(syn_power) & np.isfinite(orig_power) + denom = orig_power[finite].sum() + return syn_power[finite].sum() / denom - 1.0 if denom else float("nan") + + +class OracleMethod: + """Returns the injected truth; its signed error is ~0 at every campaign length.""" + + def __init__(self, original_df: pd.DataFrame, *, name: str = "oracle") -> None: + """Store the no-upgrade baseline used to recover the injected truth.""" + self._original = original_df + self.name = name + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Return the injected truth as the P50 estimate.""" + return MethodOutput(p50_overall=_oracle_overall_uplift(mi, self._original)) + + +class BiasedMethod: + """Returns the truth plus a fixed offset; its error sits at that offset.""" + + def __init__(self, original_df: pd.DataFrame, *, offset: float, name: str = "biased") -> None: + """Store the baseline and the fixed offset added to every estimate.""" + self._original = original_df + self._offset = offset + self.name = name + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Return the injected truth plus the fixed offset.""" + return MethodOutput(p50_overall=_oracle_overall_uplift(mi, self._original) + self._offset) + + +class ShrinkingNoiseMethod: + """Returns the truth plus noise whose size shrinks with campaign length. + + A crude stand-in for a real estimator: more campaign data -> a tighter estimate. The noise + sigma scales as ``base_sigma / sqrt(campaign_months)``, and the draw is seeded + deterministically from ``(test turbine, treatment start, campaign length)`` so each + replicate gets its own value while the whole study stays reproducible. Across the ensemble + this makes the plotted spread band narrow as the campaign grows. + """ + + def __init__(self, original_df: pd.DataFrame, *, base_sigma: float = 0.03, name: str = "realistic") -> None: + """Store the baseline and the base noise level scaled by campaign length.""" + self._original = original_df + self._base_sigma = base_sigma + self.name = name + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Return the truth plus a deterministic, campaign-length-shrinking noise draw.""" + truth = _oracle_overall_uplift(mi, self._original) + treatment_start = mi.upgrade_timing # prepost: a Timestamp + post_days = max((mi.scada_df.index.max() - treatment_start).days, 1) + months = post_days / 30.4 + sigma = self._base_sigma / np.sqrt(months) + wtg_term = sum(ord(c) for c in mi.test_wtg) * 1000 + round(months) + seed = (int(pd.Timestamp(treatment_start).value) % (2**32)) ^ wtg_term + rng = np.random.default_rng(seed) + return MethodOutput(p50_overall=truth + float(rng.normal(0.0, sigma))) + + +def run_example_study( + scada_df: pd.DataFrame, + *, + out_root: str | Path | None = None, + turbine_subset: list[str] | None = None, + treatment_start_range: tuple[pd.Timestamp, pd.Timestamp] | None = None, + min_pre_months: int = 12, + campaign_months: list[int] | None = None, + n_replicates: int = 6, + delta: float = 0.05, + seed: int = 0, +) -> pd.DataFrame: + """Score the illustrative methods on a constant-Cp study and save inspectable outputs. + + :param scada_df: wind-up-format real SCADA (all subset turbines), the no-upgrade baseline + :param out_root: output directory; defaults to :func:`default_output_root` + :param turbine_subset: turbines kept in the study (one drawn as test per replicate) + :param treatment_start_range: ``(earliest, latest)`` changeover to draw from + :param min_pre_months: fixed baseline length before the changeover, in months + :param campaign_months: the campaign-length sweep grid, in months + :param n_replicates: ensemble size (drives the spread estimate) + :param delta: the injected constant-Cp uplift fraction + :param seed: top-level seed for reproducibility + :return: the leaderboard summary frame + """ + out_dir = Path(out_root) if out_root is not None else default_output_root() + out_dir.mkdir(parents=True, exist_ok=True) + turbine_subset = turbine_subset if turbine_subset is not None else DEFAULT_TURBINE_SUBSET + if treatment_start_range is None: + treatment_start_range = (pd.Timestamp("2017-01-01", tz="UTC"), pd.Timestamp("2017-01-15", tz="UTC")) + campaign_months = campaign_months if campaign_months is not None else [3, 6, 9, 12] + + study = StudyConfig( + mode="prepost", + turbine_subset=turbine_subset, + treatment_start_range=treatment_start_range, + min_pre_months=min_pre_months, + campaign_months=campaign_months, + n_replicates=n_replicates, + seed=seed, + ) + methods: list[Method] = [ + OracleMethod(scada_df), + BiasedMethod(scada_df, offset=0.02), + ShrinkingNoiseMethod(scada_df, base_sigma=0.03), + ] + + profile_name = f"constant_cp_{delta:.0%}".replace("%", "pct") + logger.info( + "Scoring %d methods over %d replicates x campaigns %s (profile %s, true uplift %+.1f%%)", + len(methods), + n_replicates, + campaign_months, + profile_name, + delta * 100, + ) + results = score_study( + scada_df, profile=[ConstantCpChange(delta=delta)], methods=methods, study=study, profile_name=profile_name + ) + summary = leaderboard(results) + + results_path = out_dir / "results_tidy.csv" + summary_path = out_dir / "leaderboard.csv" + plot_path = out_dir / "campaign_curves.png" + results.to_csv(results_path, index=False) + summary.to_csv(summary_path, index=False) + plot_campaign_curves( + summary, + save_path=plot_path, + title=f"Hill of Towie {profile_name} (Cp delta {delta:+.1%}, {n_replicates} replicates)", + ) + + logger.info("Saved leaderboard -> %s", summary_path) + logger.info("Saved tidy results -> %s", results_path) + logger.info("Saved campaign-length curve -> %s", plot_path) + return summary + + +def main( + *, + out_root: str | Path | None = None, + data_dir: str | Path | None = None, + start_dt: pd.Timestamp = DEFAULT_START_DT, + end_dt_excl: pd.Timestamp = DEFAULT_END_DT_EXCL, + wtg_numbers: list[int] | None = None, +) -> pd.DataFrame: + """Run the harness end-to-end on real Hill of Towie data and save inspectable outputs. + + Downloads (and caches) the open Hill of Towie SCADA for a stable, no-upgrade window, then + scores the illustrative methods over a campaign-length sweep and writes the leaderboard, + the tidy results and the campaign-length curve under ``out_root``. + + :param out_root: output directory; defaults to :func:`default_output_root` + :param data_dir: Hill of Towie data/cache dir; defaults to the package default + :param start_dt: inclusive UTC window start + :param end_dt_excl: exclusive UTC window end + :param wtg_numbers: turbine numbers to load; defaults to the stable south-west cluster + :return: the leaderboard summary frame + """ + wtg_numbers = wtg_numbers if wtg_numbers is not None else DEFAULT_WTG_NUMBERS + logger.info("Loading Hill of Towie SCADA %s..%s for turbines %s", start_dt, end_dt_excl, wtg_numbers) + scada_df, _metadata_df = load_hot_scada( + start_dt=start_dt, + end_dt_excl=end_dt_excl, + wtg_numbers=wtg_numbers, + data_dir=Path(data_dir) if data_dir is not None else None, + ) + summary = run_example_study(scada_df, out_root=out_root) + logger.info("Leaderboard:\n%s", summary.to_string(index=False)) + return summary + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(name)s: %(message)s") + main() diff --git a/benchmarking/harness/leaderboard.py b/benchmarking/harness/leaderboard.py new file mode 100644 index 00000000..1762ad8c --- /dev/null +++ b/benchmarking/harness/leaderboard.py @@ -0,0 +1,107 @@ +"""Leaderboard: summarise tidy scoring results into a side-by-side comparison. + +Groups the tidy results by ``(method, profile, )`` and reduces each group's +signed errors to bias, spread and the combined RMSE score via :func:`metrics.summarize_errors`. +A pure consumer of the results table; lower ``score`` is better. + +The campaign-length column is ``campaign_months`` by default and ``campaign_weeks`` for a weeks-grid +study (see :class:`~benchmarking.harness.replicates.StudyConfig`); pass ``length_col`` to match the +results being summarised. +""" + +from __future__ import annotations + +import pandas as pd + +from benchmarking.harness.metrics import summarize_errors + +DEFAULT_LENGTH_COL = "campaign_months" + + +def _group_keys(length_col: str) -> list[str]: + return ["method", "profile", length_col] + + +def _condition_group_keys(length_col: str) -> list[str]: + return ["method", "profile", length_col, "condition", "condition_bin"] + + +def leaderboard(results_df: pd.DataFrame, *, length_col: str = DEFAULT_LENGTH_COL) -> pd.DataFrame: + """Summarise scoring results into per-(method, profile, campaign-length) bias/spread/score. + + Only overall-uplift rows (``condition == "overall"``) are summarised; per-condition rows are + excluded. Returns one row per group with ``bias``, ``spread``, ``score``, the mean recovered + and true uplift (``mean_estimate`` / ``mean_truth``, when those columns are present in the + input), ``n_replicates``, and the group's wall time (``wall_time_s_sum`` total and + ``wall_time_s_mean`` per run, when ``wall_time_s`` is present), sorted by method, profile then + campaign length. + + :param length_col: the campaign-length column in ``results_df`` (``campaign_months`` / + ``campaign_weeks``) + """ + group_keys = _group_keys(length_col) + overall = results_df[results_df["condition"] == "overall"] if "condition" in results_df else results_df + + records = [] + for keys, group in overall.groupby(group_keys, sort=True): + summary = summarize_errors(group["signed_error"].to_numpy()) + records.append( + { + **dict(zip(group_keys, keys, strict=True)), + "bias": summary.bias, + "spread": summary.spread, + "score": summary.score, + "mean_estimate": float(group["estimate"].mean()) if "estimate" in group else float("nan"), + "mean_truth": float(group["truth"].mean()) if "truth" in group else float("nan"), + "n_replicates": summary.n, + "wall_time_s_sum": float(group["wall_time_s"].sum(min_count=1)) + if "wall_time_s" in group + else float("nan"), + "wall_time_s_mean": float(group["wall_time_s"].mean()) if "wall_time_s" in group else float("nan"), + } + ) + columns = [ + *group_keys, + "bias", + "spread", + "score", + "mean_estimate", + "mean_truth", + "n_replicates", + "wall_time_s_sum", + "wall_time_s_mean", + ] + return pd.DataFrame(records, columns=columns) + + +def conditional_leaderboard(results_df: pd.DataFrame, *, length_col: str = DEFAULT_LENGTH_COL) -> pd.DataFrame: + """Per-(method, profile, campaign, condition, bin) bias/spread/score over the conditional rows. + + Every per-condition axis (``ws``, ``ti``, ``power``, …) is summarised; only overall rows + (``condition == "overall"``) are excluded. Returns one row per group with ``bias``, + ``spread``, ``score``, the mean recovered and true uplift (``mean_estimate`` / + ``mean_truth``, when those columns are present in the input), and ``n_replicates``, + sorted by method, profile, campaign length, condition, then condition_bin. + + :param length_col: the campaign-length column in ``results_df`` (``campaign_months`` / + ``campaign_weeks``) + """ + group_keys = _condition_group_keys(length_col) + cond = results_df[results_df["condition"] != "overall"] if "condition" in results_df else results_df.iloc[:0] + + records = [] + for keys, group in cond.groupby(group_keys, sort=True): + summary = summarize_errors(group["signed_error"].to_numpy()) + records.append( + { + **dict(zip(group_keys, keys, strict=True)), + "bias": summary.bias, + "spread": summary.spread, + "score": summary.score, + "mean_estimate": float(group["estimate"].mean()) if "estimate" in group else float("nan"), + "mean_truth": float(group["truth"].mean()) if "truth" in group else float("nan"), + "n_replicates": summary.n, + } + ) + columns = [*group_keys, "bias", "spread", "score", "mean_estimate", "mean_truth", "n_replicates"] + return pd.DataFrame(records, columns=columns) diff --git a/benchmarking/harness/method.py b/benchmarking/harness/method.py new file mode 100644 index 00000000..2b69e41c --- /dev/null +++ b/benchmarking/harness/method.py @@ -0,0 +1,89 @@ +"""The thin method seam the harness scores against. + +Deliberately minimal: the harness only ever sees ``MethodInput -> MethodOutput``. The mode is +inferred from ``upgrade_timing``'s type (a timestamp is prepost; a ``ToggleSchedule`` is +toggle). Reference selection and method-specific config are baked into each ``Method``, not +carried on the input, so the seam bakes in no method assumptions. The Issue 4 data contract +later enriches this behind the same seam. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING, Protocol, runtime_checkable + +if TYPE_CHECKING: + import pandas as pd + + from benchmarking.synthetic import ToggleSchedule + + +@dataclass +class MethodInput: + """The windowed data a method sees for one campaign. + + The frame carries **source-native** SCADA column names (the real tag names of the data + source). Identifying turbines is a property of the data, not a method choice, so the seam + names the turbine-identifier column here; methods read any value columns (e.g. active power) + by their own source-native config. + + :param scada_df: windowed synthetic SCADA (all subset turbines, baseline + activity) + :param test_wtg: the upgraded turbine to estimate + :param upgrade_timing: changeover timestamp (prepost); a periodic ``ToggleSchedule`` or an + explicit, possibly irregular ``toggle_df`` (a ``pd.DataFrame`` with boolean + ``toggle_on``/``toggle_off`` on a ``DatetimeIndex``) for toggle. See + :mod:`benchmarking.harness.toggle`. + :param turbine_col: the turbine-identifier column in ``scada_df`` + """ + + scada_df: pd.DataFrame + test_wtg: str + upgrade_timing: pd.Timestamp | ToggleSchedule | pd.DataFrame + turbine_col: str = "TurbineName" + + +@dataclass +class MethodOutput: + """A method's P50 uplift estimate, and optionally its uncertainty. + + The uncertainty fields are **additive**: they default to ``None``, so a method that reports + only a P50 is unchanged and the harness records a NaN uncertainty for it. + + :param p50_overall: overall P50 uplift (energy-ratio fraction) + :param p50_by_condition: optional per-condition estimates (columns ``condition``, + ``condition_bin``, ``p50_uplift``); ``condition`` ∈ {"ws","ti","power"}; ``None`` when the + method produces only an overall number. May also carry a ``sigma_uplift`` column, the + per-bin counterpart of ``sigma_overall``. + :param sigma_overall: optional 1-sigma on ``p50_overall``, a symmetric delta in energy-ratio + fraction. Scored against the deviation from ground truth, so it is a *total* uncertainty. + :param uncertainty_diagnostics: optional tidy frame keyed by ``(condition, condition_bin)`` + carrying whatever a method wants to say about how its uncertainty was reached. The harness + never interprets these columns; it merges them onto the results rows and carries them + through. Use ``("overall", "overall")`` for the headline row. + :param labeled_rows: optional per-record frame exposing the row selection the estimate was + actually built from: the test turbine's original SCADA columns, plus ``used`` (did the row + survive the method's filtering), ``segment`` (``baseline`` / ``upgraded`` / ``excluded``), + and one ``_bin`` column per condition the method was asked for. It lets a + consumer compute a per-bin quantity the method does not itself report -- a mean pitch, say + -- over exactly the rows and bins the uplift used, instead of re-deriving the filtering and + binning and hoping the two agree. Deliberately row-level rather than pre-aggregated, so it + serves consumers whose desired aggregate is not known here. The harness never interprets + it. + """ + + p50_overall: float + p50_by_condition: pd.DataFrame | None = None + sigma_overall: float | None = None + uncertainty_diagnostics: pd.DataFrame | None = None + labeled_rows: pd.DataFrame | None = None + + +@runtime_checkable +class Method(Protocol): + """A pluggable uplift estimator: a name plus ``estimate(MethodInput) -> MethodOutput``.""" + + name: str + + def estimate(self, mi: MethodInput) -> MethodOutput: + """Return a P50 uplift estimate for the campaign described by ``mi``.""" + ... diff --git a/benchmarking/harness/metrics.py b/benchmarking/harness/metrics.py new file mode 100644 index 00000000..d8bb83f2 --- /dev/null +++ b/benchmarking/harness/metrics.py @@ -0,0 +1,46 @@ +"""Accuracy, precision and combined score over a set of signed errors. + +A method's signed error on one campaign is ``estimate - truth``. Aggregated across an +ensemble of replicates these give: + +- **accuracy / bias** = mean signed error, +- **precision / spread** = population standard deviation (``ddof=0``), +- **combined score** = RMSE of the signed errors = ``sqrt(mean(error**2))``. + +With the population spread the score is exactly ``sqrt(bias**2 + spread**2)``, so it is small +only when *both* accuracy and precision are good. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +import numpy as np +import numpy.typing as npt + + +@dataclass(frozen=True) +class ErrorSummary: + """Bias, spread and combined score over ``n`` finite signed errors.""" + + bias: float + spread: float + score: float + n: int + + +def summarize_errors(signed_errors: npt.ArrayLike) -> ErrorSummary: + """Summarise signed errors into bias, spread (population std) and RMSE score. + + Non-finite errors (e.g. a replicate that produced no estimate) are dropped before + aggregating. An empty (or all-NaN) input yields a NaN summary with ``n == 0``. + """ + errors = np.asarray(signed_errors, dtype=float).ravel() + errors = errors[np.isfinite(errors)] + n = int(errors.size) + if n == 0: + return ErrorSummary(bias=float("nan"), spread=float("nan"), score=float("nan"), n=0) + bias = float(errors.mean()) + spread = float(errors.std(ddof=0)) + score = float(np.sqrt(np.mean(errors**2))) + return ErrorSummary(bias=bias, spread=spread, score=score, n=n) diff --git a/benchmarking/harness/plots.py b/benchmarking/harness/plots.py new file mode 100644 index 00000000..b5d0df91 --- /dev/null +++ b/benchmarking/harness/plots.py @@ -0,0 +1,151 @@ +"""Campaign-length curves for comparing methods. + +Three stacked panels sharing one campaign-length x-axis: + +1. **Uplift recovery** — the true (injected) uplift plus each method's mean recovered uplift, + so over- vs under-estimation is visible and the true uplift's variation with campaign + length is shown. +2. **Bias +/- spread** — one ``bias`` line per method with a shaded ``bias +/- spread`` band + (accuracy and precision together). +3. **Score** — the combined ``score`` (RMSE) per method. + +All quantities are converted from uplift fractions to percentage points (a 0.01 fraction is +plotted as 1 pp). Each method keeps one colour across all panels. Consumes the leaderboard +summary frame. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np + +from benchmarking.harness.leaderboard import DEFAULT_LENGTH_COL + +if TYPE_CHECKING: + from pathlib import Path + + import pandas as pd + from matplotlib.figure import Figure + +# Uplift quantities are stored as fractions; plot them as percentage points. +_FRACTION_TO_PP = 100.0 + + +def plot_campaign_curves( + summary_df: pd.DataFrame, + *, + save_path: str | Path | None = None, + title: str | None = None, + length_col: str = DEFAULT_LENGTH_COL, +) -> Figure: + """Plot uplift recovery, bias/spread and score vs campaign length, per method. + + :param summary_df: a leaderboard summary (columns ``method``, the campaign-length column, + ``bias``, ``spread``, ``score``, and optionally ``mean_estimate`` / ``mean_truth`` for the + top panel) + :param save_path: if given, the figure is written here (PNG) + :param title: optional title for the top panel + :param length_col: the campaign-length column in ``summary_df`` (``campaign_months`` / + ``campaign_weeks``); also names the x axis + :return: the matplotlib Figure + """ + methods = sorted(summary_df["method"].unique()) + colors = {method: f"C{i}" for i, method in enumerate(methods)} + + fig, (ax_uplift, ax_band, ax_score) = plt.subplots(3, 1, sharex=True, figsize=(8, 11)) + + # Top panel: the true uplift (method-independent) and each method's mean recovered uplift. + if "mean_truth" in summary_df.columns: + truth = summary_df.groupby(length_col)["mean_truth"].mean().sort_index() + ax_uplift.plot( + truth.index.to_numpy(), truth.to_numpy() * _FRACTION_TO_PP, "--", marker="s", color="k", label="true uplift" + ) + + for method in methods: + group = summary_df[summary_df["method"] == method].sort_values(length_col) + lengths = group[length_col].to_numpy() + color = colors[method] + bias = group["bias"].to_numpy() * _FRACTION_TO_PP + spread = group["spread"].to_numpy() * _FRACTION_TO_PP + + if "mean_estimate" in summary_df.columns: + estimate = group["mean_estimate"].to_numpy() * _FRACTION_TO_PP + ax_uplift.plot(lengths, estimate, marker="o", color=color, label=method) + ax_uplift.fill_between(lengths, estimate - spread, estimate + spread, alpha=0.15, color=color) + ax_band.plot(lengths, bias, marker="o", color=color, label=method) + ax_band.fill_between(lengths, bias - spread, bias + spread, alpha=0.15, color=color) + ax_score.plot(lengths, group["score"].to_numpy() * _FRACTION_TO_PP, marker="o", color=color, label=method) + + ax_uplift.set_ylabel("Measured uplift [pp]") + ax_uplift.set_title(title if title is not None else "P50 uplift recovery vs campaign length") + ax_uplift.grid(visible=True, alpha=0.3) + ax_uplift.legend() + + ax_band.axhline(0.0, color="k", linewidth=0.8) + ax_band.set_ylabel("Bias +/- spread [pp]") + ax_band.grid(visible=True, alpha=0.3) + + ax_score.set_xlabel(f"Campaign length [{length_col.removeprefix('campaign_')}]") + ax_score.set_ylabel("Score / RMSE [pp]") + ax_score.grid(visible=True, alpha=0.3) + ax_score.set_xlim(left=0.0) + _set_score_ylim(ax_score, summary_df["score"].to_numpy() * _FRACTION_TO_PP) + + fig.tight_layout() + + if save_path is not None: + fig.savefig(save_path, dpi=150) + return fig + + +def _set_score_ylim(ax: plt.Axes, scores_pp: np.ndarray) -> None: + """Anchor the score y-axis at min(0, lowest point) minus a small margin. + + Score is a non-negative RMSE, but a hard floor of 0 clips data points sitting exactly at 0 + (e.g. an oracle). Drop the floor by a small margin so those points stay visible. + """ + lo = float(np.nanmin(scores_pp)) + hi = float(np.nanmax(scores_pp)) + span = hi - lo + margin = 0.05 * span if span > 0 else max(abs(hi), 1.0) * 0.05 + ax.set_ylim(bottom=min(0.0, lo) - margin) + + +def plot_conditional_uplift( + summary_df: pd.DataFrame, + *, + condition: str, + save_path: str | Path | None = None, + title: str | None = None, +) -> Figure: + """Plot mean recovered vs true uplift across bins of one condition, with a bias±spread band.""" + df = summary_df[summary_df["condition"] == condition].copy() + df["_left"] = df["condition_bin"].str.extract(r"\(([-0-9.]+),").astype(float) + df = df.sort_values("_left") + order = df.drop_duplicates("condition_bin")["condition_bin"].tolist() + x = np.arange(len(order)) + + fig, ax = plt.subplots(figsize=(9, 5)) + truth = df.drop_duplicates("condition_bin").set_index("condition_bin").reindex(order)["mean_truth"] + ax.plot(x, truth.to_numpy() * _FRACTION_TO_PP, "--", marker="s", color="k", label="true uplift") + for i, method in enumerate(sorted(df["method"].unique())): + m = df[df["method"] == method].set_index("condition_bin").reindex(order) + est = m["mean_estimate"].to_numpy() * _FRACTION_TO_PP + ax.plot(x, est, marker="o", color=f"C{i}", label=method) + if "spread" in m: + sp = m["spread"].to_numpy() * _FRACTION_TO_PP + ax.fill_between(x, est - sp, est + sp, color=f"C{i}", alpha=0.2) + ax.set_xticks(x) + ax.set_xticklabels(order, rotation=45, ha="right") + ax.set_xlabel(condition) + ax.set_ylabel("uplift [pp]") + ax.axhline(0.0, color="grey", lw=0.8) + ax.legend() + if title: + ax.set_title(title) + fig.tight_layout() + if save_path is not None: + fig.savefig(save_path, dpi=120) + return fig diff --git a/benchmarking/harness/replicates.py b/benchmarking/harness/replicates.py new file mode 100644 index 00000000..6a808080 --- /dev/null +++ b/benchmarking/harness/replicates.py @@ -0,0 +1,204 @@ +"""The replicate ensemble — the precision axis of the harness. + +Precision needs an ensemble: a single dataset gives one estimate, hence one error. Holding +the upgrade *profile* fixed, an ensemble is built by varying the base ingredients — the test +turbine and the treatment-start date (the changeover for prepost; when toggling begins for +toggle). The spread of a method's error across these replicates is its precision. + +The data is first subset to ``turbine_subset`` (one test turbine is drawn per replicate; the +rest are its references) so each run stays light — important when scoring many replicates. +Draws are a pure deterministic function of ``(StudyConfig, seed)``. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import TYPE_CHECKING, Literal + +import numpy as np + +from benchmarking.harness.campaign import resolve_campaign_grid +from benchmarking.synthetic import HOT_COLUMNS, ToggleSchedule, generate_dataset + +if TYPE_CHECKING: + from collections.abc import Iterator + + import numpy.typing as npt + import pandas as pd + + from benchmarking.harness.campaign import CampaignUnit + from benchmarking.synthetic import ColumnSchema, SyntheticDataset, UpliftResult + + +@dataclass(frozen=True) +class StudyConfig: + """One study: a profile evaluated over an ensemble and a campaign-length grid. + + The campaign-length grid is given in **either** months or weeks — exactly one of + ``campaign_months`` / ``campaign_weeks`` must be set. Months is the original grid (used by the + overnight studies and their frozen baselines); weeks suits short campaigns, where a month is too + coarse a step. Read the grid generically via :attr:`campaign_lengths` / :attr:`campaign_unit` / + :attr:`campaign_length_col` rather than touching the two fields directly. + + :param mode: ``"prepost"`` or ``"toggle"`` + :param turbine_subset: the only turbines kept in the data; per replicate one is drawn as + the test turbine and the rest are its references + :param treatment_start_range: ``(earliest, latest)`` treatment-start timestamp to draw from + :param min_pre_months: fixed baseline length before treatment start, in months (always months, + independent of the campaign grid's unit) + :param n_replicates: number of ``(turbine, treatment_start)`` instances to draw + :param campaign_months: the campaign-length sweep grid, in months (exclusive with ``campaign_weeks``) + :param campaign_weeks: the campaign-length sweep grid, in weeks (exclusive with ``campaign_months``) + :param toggle_period: toggle on/off cycle length (toggle mode only) + :param seed: RNG seed for the draws + """ + + mode: Literal["prepost", "toggle"] + turbine_subset: list[str] + treatment_start_range: tuple[pd.Timestamp, pd.Timestamp] + min_pre_months: int + n_replicates: int + campaign_months: list[int] | None = None + campaign_weeks: list[int] | None = None + toggle_period: pd.Timedelta | None = None + seed: int = 0 + + def __post_init__(self) -> None: + """Validate that exactly one campaign-length grid is set.""" + resolve_campaign_grid(campaign_months=self.campaign_months, campaign_weeks=self.campaign_weeks) + + @property + def campaign_lengths(self) -> list[int]: + """The campaign-length grid values, whichever unit they are in.""" + return resolve_campaign_grid(campaign_months=self.campaign_months, campaign_weeks=self.campaign_weeks)[0] + + @property + def campaign_unit(self) -> CampaignUnit: + """The campaign grid's unit: ``"months"`` or ``"weeks"``.""" + return resolve_campaign_grid(campaign_months=self.campaign_months, campaign_weeks=self.campaign_weeks)[1] + + @property + def campaign_length_col(self) -> str: + """The result column campaign lengths are reported under (``campaign_months``/``_weeks``).""" + return f"campaign_{self.campaign_unit}" + + @property + def max_activity_months(self) -> int: + """The longest campaign the study scores, in months. Raises for a weeks study. + + Raises rather than converting so a weeks study reaching a months-only consumer fails loudly + instead of silently reporting a week count as a month count. + """ + if self.campaign_unit != "months": + msg = ( + "max_activity_months is only defined for a months grid, but this study has " + f"campaign_unit={self.campaign_unit!r}; read campaign_lengths instead" + ) + raise ValueError(msg) + return max(self.campaign_lengths) + + +@dataclass +class Replicate: + """One generated dataset paired with the ingredients that produced it.""" + + dataset: SyntheticDataset + test_wtg: str + treatment_start: pd.Timestamp + upgrade_timing: pd.Timestamp | ToggleSchedule + replicate_id: int = field(default=0) + + @property + def synthetic_df(self) -> pd.DataFrame: + """The method-facing synthetic SCADA (all subset turbines).""" + return self.dataset.synthetic_df + + def true_uplift(self, **kwargs: object) -> UpliftResult: + """Ground-truth uplift for this replicate's test turbine (delegates to the dataset).""" + kwargs.setdefault("test_wtg", self.test_wtg) + return self.dataset.true_uplift(**kwargs) # type: ignore[arg-type] + + +def iter_replicates( + base_scada: pd.DataFrame, + *, + profile: list, + study: StudyConfig, + columns: ColumnSchema = HOT_COLUMNS, +) -> Iterator[Replicate]: + """Yield ``study.n_replicates`` replicates of ``profile`` one at a time. + + The streaming counterpart of :func:`build_replicates`, yielding the same replicates in the same + order. A replicate carries a ``synthetic_df`` *and* the ``original_df`` its ground truth needs + (~0.5 GB for a multi-year, few-turbine dataset), so a large ensemble must iterate here and let + each be freed rather than materialising them all. + + :param columns: the source-native column schema ``base_scada`` is keyed by + """ + subset = base_scada[base_scada[columns.turbine].isin(study.turbine_subset)] + candidates = _candidate_starts(subset.index, study.treatment_start_range) + + rng = np.random.default_rng(study.seed) + turbines = np.asarray(study.turbine_subset) + for replicate_id in range(study.n_replicates): + test_wtg = str(rng.choice(turbines)) + treatment_start = candidates[int(rng.integers(len(candidates)))] + upgrade_timing = _upgrade_timing(study, treatment_start) + dataset = generate_dataset( + scada_df=subset, + test_wtgs=[test_wtg], + upgrades=profile, + mode=study.mode, + upgrade_timing=upgrade_timing, + columns=columns, + seed=study.seed, + ) + yield Replicate( + dataset=dataset, + test_wtg=test_wtg, + treatment_start=treatment_start, + upgrade_timing=upgrade_timing, + replicate_id=replicate_id, + ) + + +def build_replicates( + base_scada: pd.DataFrame, + *, + profile: list, + study: StudyConfig, + columns: ColumnSchema = HOT_COLUMNS, +) -> list[Replicate]: + """Draw ``study.n_replicates`` replicates of ``profile`` from ``base_scada``. + + Subsets the data to ``turbine_subset``, then draws ``(test turbine, treatment_start)`` pairs + deterministically from ``seed`` and injects the profile via the synthetic generator. + + Materialises every replicate at once; see :func:`iter_replicates` to stream them instead when + the ensemble is large enough for that to matter. + + :param columns: the source-native column schema ``base_scada`` is keyed by + """ + return list(iter_replicates(base_scada, profile=profile, study=study, columns=columns)) + + +def _candidate_starts( + index: pd.DatetimeIndex, treatment_start_range: tuple[pd.Timestamp, pd.Timestamp] +) -> npt.NDArray[np.datetime64]: + """Return unique on-grid timestamps within the draw range (so draws land on real records).""" + lo, hi = treatment_start_range + unique = index.unique() + candidates = unique[(unique >= lo) & (unique <= hi)] + if len(candidates) == 0: + msg = f"no records in treatment_start_range {treatment_start_range}" + raise ValueError(msg) + return candidates.sort_values().to_numpy() + + +def _upgrade_timing(study: StudyConfig, treatment_start: pd.Timestamp) -> pd.Timestamp | ToggleSchedule: + if study.mode == "toggle": + if study.toggle_period is None: + msg = "toggle_period is required for mode='toggle'" + raise ValueError(msg) + return ToggleSchedule(period=study.toggle_period, start=treatment_start) + return treatment_start diff --git a/benchmarking/harness/scoring.py b/benchmarking/harness/scoring.py new file mode 100644 index 00000000..bc39a8fa --- /dev/null +++ b/benchmarking/harness/scoring.py @@ -0,0 +1,296 @@ +"""The scoring orchestrator: run methods over an ensemble and a campaign-length grid. + +For one study it builds the replicate ensemble, materialises the ``(replicate, campaign)`` +instances **once**, and scores every method against that identical list. Two fairness +properties follow structurally: + +1. **Across methods** — instances are built before any method runs, so every method sees the + exact same ``MethodInput``s; only the estimate differs. +2. **Across campaign lengths** — all lengths of one replicate share its + ``(turbine, baseline_start, treatment_start)`` and differ only in ``activity_end``, so + shorter windows are leading prefixes of longer ones (a property of ``campaign_windows``). + +For each instance the method's estimate and the ground truth are computed over the **same +records**, so the signed error is exact at every campaign length. Output is one tidy +long-format DataFrame, the input to the leaderboard and plots. + +:func:`score_one` is the per-``(method, instance)`` unit :func:`score_study` is built from, public so +a caller that cannot afford the materialised ensemble (a large replicate ensemble would exhaust +memory) can drive its own replicate-outer loop over :func:`iter_replicates` without reimplementing +the truth alignment. +""" + +from __future__ import annotations + +import time +from typing import TYPE_CHECKING + +import pandas as pd + +from benchmarking.harness.campaign import campaign_windows, treated_activity_mask, window_row_mask +from benchmarking.harness.conditions import condition_bins +from benchmarking.harness.method import MethodInput +from benchmarking.harness.replicates import build_replicates +from benchmarking.synthetic import HOT_COLUMNS + +if TYPE_CHECKING: + from collections.abc import Callable + + import numpy as np + import numpy.typing as npt + + from benchmarking.harness.campaign import CampaignWindow + from benchmarking.harness.method import Method + from benchmarking.harness.replicates import Replicate, StudyConfig + from benchmarking.synthetic import ColumnSchema + +# The columns a method's ``uncertainty_diagnostics`` keys on, and therefore does not contribute. +_DIAGNOSTIC_KEYS = ("condition", "condition_bin") +# Columns the harness itself writes on every result row. A method's diagnostics may not use these +# names — see :func:`_merge_diagnostics`. The campaign-length column is study-dependent +# (``campaign_months`` / ``campaign_weeks``), so both are reserved regardless of the study's unit. +_RESERVED_COLUMNS = frozenset( + { + "method", + "profile", + "replicate", + "test_wtg", + "campaign_months", + "campaign_weeks", + "treatment_start", + "baseline_start", + "activity_end", + "condition", + "condition_bin", + "estimate", + "truth", + "signed_error", + "sigma", + "wall_time_s", + } +) + + +def score_study( + base_scada: pd.DataFrame, + *, + profile: list, + methods: list[Method], + study: StudyConfig, + profile_name: str = "profile", + columns: ColumnSchema = HOT_COLUMNS, + on_method_complete: Callable[[str, pd.DataFrame], None] | None = None, +) -> pd.DataFrame: + """Score ``methods`` on ``study`` over ``profile`` injected into ``base_scada``. + + Returns a tidy long-format frame: one row per method x replicate x campaign length, with + the P50 ``estimate``, the ground-truth ``truth`` and their ``signed_error``. Each row also + carries the window it was tested over — ``treatment_start`` (the upgrade start), + ``baseline_start`` and ``activity_end`` — so a result is self-describing. + + :param columns: the source-native column schema ``base_scada`` is keyed by + :param on_method_complete: optional hook called as each method finishes its full instance + sweep, with ``(method_name, that_method's_rows)`` (the same rows it contributes to the + returned frame). Lets a caller act on a method's results early — e.g. plot them — instead + of waiting for every method, useful when a slow method runs last. It never changes the + returned frame; order methods fastest-first to get the earliest feedback. + """ + replicates = build_replicates(base_scada, profile=profile, study=study, columns=columns) + data_start = base_scada.index.min() + data_end = base_scada.index.max() + instances = _materialise_instances(replicates, study, data_start=data_start, data_end=data_end) + # Truth depends only on ``(replicate, window)``, so compute it once here rather than once + # per method (it would otherwise carry an avoidable ``len(methods)`` multiplier). + truth_masks = [truth_mask(r, w) for r, w in instances] + truths = [r.true_uplift(mask=m).overall for (r, _), m in zip(instances, truth_masks, strict=True)] + + rows = [] + for method in methods: + method_rows: list[dict[str, object]] = [] + for (replicate, window), truth, mask in zip(instances, truths, truth_masks, strict=True): + method_rows.extend( + score_one(method, replicate=replicate, window=window, truth=truth, mask=mask, profile_name=profile_name) + ) + if on_method_complete is not None: + on_method_complete(method.name, pd.DataFrame(method_rows)) + rows.extend(method_rows) + return pd.DataFrame(rows) + + +def score_one( + method: Method, + *, + replicate: Replicate, + window: CampaignWindow, + truth: float, + mask: npt.NDArray[np.bool_], + profile_name: str = "profile", +) -> list[dict[str, object]]: + """Run one method over one ``(replicate, window)`` instance and return its tidy result rows. + + The overall row plus one per condition bin the method reports, each with the estimate, ``truth``, + their signed error and the method's ``sigma`` (NaN when it reports no uncertainty). + + ``truth`` and ``mask`` are passed in because they depend only on ``(replicate, window)``; + deriving them here would multiply that cost by ``len(methods)``. Build them with + :func:`truth_mask` and ``replicate.true_uplift(mask=...)``. + """ + method_input = _method_input(replicate, window) + start = time.perf_counter() + output = method.estimate(method_input) + wall_time_s = time.perf_counter() - start + base_fields: dict[str, object] = { + "method": method.name, + "profile": profile_name, + "replicate": replicate.replicate_id, + "test_wtg": replicate.test_wtg, + window.length_col: window.length, + "treatment_start": window.treatment_start, + "baseline_start": window.baseline_start, + "activity_end": window.activity_end, + } + rows: list[dict[str, object]] = [ + { + **base_fields, + "condition": "overall", + "condition_bin": "overall", + "estimate": output.p50_overall, + "truth": truth, + "signed_error": output.p50_overall - truth, + "sigma": _optional_float(output.sigma_overall), + "wall_time_s": wall_time_s, + } + ] + if output.p50_by_condition is not None: + rows.extend(_conditional_rows(output.p50_by_condition, replicate, window, mask, base_fields)) + _merge_diagnostics(rows, output.uncertainty_diagnostics) + return rows + + +def _optional_float(value: float | None) -> float: + """Return ``value`` as a float, with ``None`` (no uncertainty reported) becoming NaN.""" + return float("nan") if value is None else float(value) + + +def _merge_diagnostics(rows: list[dict[str, object]], diagnostics: pd.DataFrame | None) -> None: + """Merge a method's uncertainty diagnostics onto ``rows`` in place, keyed by condition and bin. + + Attached without interpretation. Rows with no matching diagnostics row are left alone. A column + colliding with one the harness owns raises rather than silently overwriting the estimate or truth + the row exists to report. + """ + if diagnostics is None: + return + missing = [c for c in _DIAGNOSTIC_KEYS if c not in diagnostics.columns] + if missing: + msg = f"uncertainty_diagnostics must be keyed by {list(_DIAGNOSTIC_KEYS)}; missing {missing}" + raise ValueError(msg) + extra_cols = [c for c in diagnostics.columns if c not in _DIAGNOSTIC_KEYS] + clashes = sorted(set(extra_cols) & _RESERVED_COLUMNS) + if clashes: + msg = ( + f"uncertainty_diagnostics columns {clashes} clash with columns the harness owns " + f"({sorted(_RESERVED_COLUMNS)}); rename them in the method" + ) + raise ValueError(msg) + keys = [(str(row["condition"]), str(row["condition_bin"])) for _, row in diagnostics.iterrows()] + duplicates = sorted({k for k in keys if keys.count(k) > 1}) + if duplicates: + msg = ( + f"uncertainty_diagnostics has duplicate {list(_DIAGNOSTIC_KEYS)} keys {duplicates}; " + f"the frame must hold one row per key" + ) + raise ValueError(msg) + by_key = { + key: {col: row[col] for col in extra_cols} for key, (_, row) in zip(keys, diagnostics.iterrows(), strict=True) + } + for row in rows: + row.update(by_key.get((str(row["condition"]), str(row["condition_bin"])), {})) + + +def _materialise_instances( + replicates: list[Replicate], + study: StudyConfig, + *, + data_start: pd.Timestamp, + data_end: pd.Timestamp, +) -> list[tuple[Replicate, CampaignWindow]]: + """Build the fixed ``(replicate, campaign window)`` list shared by every method.""" + instances: list[tuple[Replicate, CampaignWindow]] = [] + for replicate in replicates: + windows = campaign_windows( + replicate.treatment_start, + min_pre_months=study.min_pre_months, + campaign_months=study.campaign_months, + campaign_weeks=study.campaign_weeks, + data_start=data_start, + data_end=data_end, + ) + instances.extend((replicate, window) for window in windows) + return instances + + +def _method_input(replicate: Replicate, window: CampaignWindow) -> MethodInput: + """Return the method-facing rows: all subset turbines within ``[baseline_start, activity_end)``.""" + synthetic = replicate.synthetic_df + row_mask = window_row_mask(synthetic.index, window) + return MethodInput( + scada_df=synthetic.loc[row_mask], + test_wtg=replicate.test_wtg, + upgrade_timing=replicate.upgrade_timing, + turbine_col=replicate.dataset.columns.turbine, + ) + + +def truth_mask(replicate: Replicate, window: CampaignWindow) -> npt.NDArray[np.bool_]: + """Return the treated-activity mask over the test turbine's rows for this window. + + The mask ``replicate.true_uplift(mask=...)`` and :func:`score_one` must agree on, so a caller + driving its own loop scores against the same records :func:`score_study` would. + """ + synthetic = replicate.synthetic_df + test_index = synthetic.loc[synthetic[replicate.dataset.columns.turbine] == replicate.test_wtg].index + return treated_activity_mask(test_index, replicate.upgrade_timing, window=window) + + +def _conditional_rows( + by_condition: pd.DataFrame, + replicate: Replicate, + window: CampaignWindow, # noqa: ARG001 (kept for symmetry / future use) + mask: npt.NDArray[np.bool_], + base_fields: dict[str, object], +) -> list[dict[str, object]]: + """Join each method per-bin estimate to per-bin truth for every condition present. + + A ``sigma_uplift`` column on ``by_condition`` is carried through as the row's ``sigma``; a + method that reports per-bin P50s without per-bin uncertainty simply omits it and reads NaN. + """ + rows: list[dict[str, object]] = [] + # power edges scale with the source's baseline rating; ws/ti ignore it (fixed edges). + rated_power_kw = float(replicate.dataset.run_metadata["rated_power_kw"]) + has_sigma = "sigma_uplift" in by_condition.columns + for condition, est in by_condition.groupby("condition"): + bins = condition_bins(str(condition), rated_power_kw=rated_power_kw) + truth_df = replicate.true_uplift(mask=mask, by=condition, bins=bins).by_condition + if truth_df is None: # always set when by= is passed; guard for type narrowing + continue # pragma: no cover + truth_series = truth_df.assign(condition_bin=truth_df["condition_bin"].astype(str)).set_index("condition_bin")[ + "true_uplift" + ] + for _, r in est.iterrows(): + bin_label = str(r["condition_bin"]) + t = float(truth_series.get(bin_label, float("nan"))) + e = float(r["p50_uplift"]) + rows.append( + { + **base_fields, + "condition": condition, + "condition_bin": bin_label, + "estimate": e, + "truth": t, + "signed_error": e - t, + "sigma": float(r["sigma_uplift"]) if has_sigma else float("nan"), + "wall_time_s": float("nan"), + } + ) + return rows diff --git a/benchmarking/harness/toggle.py b/benchmarking/harness/toggle.py new file mode 100644 index 00000000..ac754d7a --- /dev/null +++ b/benchmarking/harness/toggle.py @@ -0,0 +1,148 @@ +"""The single, shared interpretation of a toggle campaign, consumed by every benchmarking method. + +A toggle campaign can be described two ways: a **periodic** ``ToggleSchedule`` (the harness's +convenience form) or an explicit, possibly **irregular** ``toggle_df`` (per-timestamp boolean +``toggle_on``/``toggle_off``, the form released wind_up already uses and a real shuffled campaign +needs). Prepost is described by a changeover ``pd.Timestamp``. + +Whatever the form, :func:`resolve_toggle` turns it into the three row-sets a method legitimately +needs, so ``naive_ratio``, ``power_model`` and ``v0_binned`` all agree on which rows are what: + +- ``upgraded`` — the "on" rows (the segment whose uplift is estimated). +- ``campaign_baseline`` — the strict "off" rows (``toggle_off``). The on/off energy comparison and + the power model's covariate matching use this. +- ``training_baseline`` — the lenient baseline ``(index < first upgraded) U toggle_off``: all + pre-campaign rows plus the off-blocks. Only the counterfactual **fit** uses this (more + upgrade-invariant data helps the model); it mirrors released wind_up's detrend-data selection. + +Rows that are neither ``upgraded`` nor a ``training_baseline`` row (e.g. a third perturbation state, +noise/ice rows, transition bins) are **excluded from every segment** — the "neither" case a binary +``treated``/``~treated`` split cannot represent. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd + +from benchmarking.synthetic import ToggleSchedule, treated_mask + +if TYPE_CHECKING: + import numpy.typing as npt + +_TOGGLE_COLUMNS = ("toggle_on", "toggle_off") + + +@dataclass(frozen=True) +class ToggleRowSets: + """The three boolean row-sets a method selects over, aligned to the index passed to resolve. + + :param upgraded: the "on" rows (uplift is estimated over these) + :param campaign_baseline: the strict "off" rows (on/off comparison + matching) + :param training_baseline: the lenient baseline (pre-campaign U off) for counterfactual fitting + """ + + upgraded: npt.NDArray[np.bool_] + campaign_baseline: npt.NDArray[np.bool_] + training_baseline: npt.NDArray[np.bool_] + + +def is_toggle(upgrade_timing: object) -> bool: + """Return whether ``upgrade_timing`` describes a toggle campaign (schedule or explicit frame). + + A ``pd.Timestamp`` (prepost changeover) is not a toggle; a ``ToggleSchedule`` or a + ``toggle_df`` (``pd.DataFrame`` with ``toggle_on``/``toggle_off``) is. + """ + return isinstance(upgrade_timing, (ToggleSchedule, pd.DataFrame)) + + +def build_toggle_df(index: pd.DatetimeIndex, schedule: ToggleSchedule) -> pd.DataFrame: + """Build the canonical three-valued ``toggle_df`` from a periodic ``ToggleSchedule``. + + Before ``schedule.start`` both flags are False (no campaign signal yet); from the start onward + exactly one is True per record (``toggle_on`` = upgraded, ``toggle_off`` = the interleaved + off-blocks). Indexed by the unique timestamps of ``index``. + """ + unique = pd.DatetimeIndex(pd.unique(index)).sort_values() + toggle_on = np.asarray(treated_mask(unique, schedule)) + after_start = np.asarray(unique >= schedule.start) if schedule.start is not None else np.ones(len(unique), bool) + toggle_off = ~toggle_on & after_start + return pd.DataFrame({"toggle_on": toggle_on, "toggle_off": toggle_off}, index=unique) + + +def _reindex_flag(toggle_df: pd.DataFrame, column: str, index: pd.DatetimeIndex) -> npt.NDArray[np.bool_]: + """Reindex one boolean ``toggle_df`` column onto ``index`` (repeats allowed), missing → False.""" + return toggle_df[column].reindex(index, fill_value=False).to_numpy(dtype=bool) + + +def _validate_toggle_df(toggle_df: pd.DataFrame) -> None: + """Validate a ``toggle_df`` before it is reindexed onto an analysis index, with clear errors. + + A frame (downstream-supplied or built here) must carry both flag columns and hold one row per + timestamp: a duplicated index would otherwise make the reindex in :func:`resolve_toggle` fail + with pandas' opaque "cannot reindex on an axis with duplicate labels". + """ + missing = [c for c in _TOGGLE_COLUMNS if c not in toggle_df.columns] + if missing: + msg = f"toggle_df must have columns {_TOGGLE_COLUMNS}; missing {missing}" + raise ValueError(msg) + if not toggle_df.index.is_unique: + dupes = toggle_df.index[toggle_df.index.duplicated()].unique().tolist() + msg = f"toggle_df index must be unique (one row per timestamp); duplicated timestamps: {dupes}" + raise ValueError(msg) + + +def resolve_toggle( + upgrade_timing: pd.Timestamp | ToggleSchedule | pd.DataFrame, index: pd.DatetimeIndex +) -> ToggleRowSets: + """Resolve any campaign description into the three row-sets, aligned to ``index``. + + ``index`` may be a repeated (long-frame) index; the labels broadcast to every row. A prepost + ``pd.Timestamp`` splits on the changeover (both baselines are the pre-changeover rows). A + ``ToggleSchedule`` is first turned into a ``toggle_df`` via :func:`build_toggle_df`. + """ + index = pd.DatetimeIndex(index) + if not is_toggle(upgrade_timing): + upgraded = np.asarray(index >= upgrade_timing) + baseline = ~upgraded + return ToggleRowSets(upgraded=upgraded, campaign_baseline=baseline, training_baseline=baseline) + + toggle_df = build_toggle_df(index, upgrade_timing) if isinstance(upgrade_timing, ToggleSchedule) else upgrade_timing + _validate_toggle_df(toggle_df) + + upgraded = _reindex_flag(toggle_df, "toggle_on", index) + campaign_baseline = _reindex_flag(toggle_df, "toggle_off", index) + if (upgraded & campaign_baseline).any(): + msg = "toggle_on and toggle_off cannot both be True for the same timestamp" + raise ValueError(msg) + + if upgraded.any(): + first_upgraded = index[upgraded].min() + pre_campaign = np.asarray(index < first_upgraded) + else: + pre_campaign = np.zeros(len(index), dtype=bool) + training_baseline = pre_campaign | campaign_baseline + return ToggleRowSets(upgraded=upgraded, campaign_baseline=campaign_baseline, training_baseline=training_baseline) + + +def toggle_upgrade_start( + upgrade_timing: pd.Timestamp | ToggleSchedule | pd.DataFrame, index: pd.DatetimeIndex +) -> pd.Timestamp: + """Return the campaign's upgrade-start timestamp, shared for run naming and time-decay weighting. + + Prepost: the changeover timestamp. A ``ToggleSchedule``: its ``start`` (or ``index`` min when + open-ended). An explicit ``toggle_df``: the first upgraded row *within* ``index`` — resolved + through :func:`resolve_toggle` so a malformed frame fails with the same clear error as the rest + of the harness, and the start is aligned to the analysis window rather than trusting on-rows the + analysis never sees — falling back to ``index`` min when the campaign never turns on inside it. + """ + if isinstance(upgrade_timing, ToggleSchedule): + return upgrade_timing.start if upgrade_timing.start is not None else pd.DatetimeIndex(index).min() + if isinstance(upgrade_timing, pd.DataFrame): + upgraded = resolve_toggle(upgrade_timing, index).upgraded + index = pd.DatetimeIndex(index) + return index[upgraded].min() if upgraded.any() else index.min() + return pd.Timestamp(upgrade_timing) diff --git a/benchmarking/synthetic/__init__.py b/benchmarking/synthetic/__init__.py new file mode 100644 index 00000000..fc4f5b79 --- /dev/null +++ b/benchmarking/synthetic/__init__.py @@ -0,0 +1,52 @@ +"""Synthetic upgrade-dataset generator (v1 benchmarking, WS1). + +Inject a known turbine upgrade into real no-upgrade SCADA to create datasets with a +derivable ground-truth uplift, for objectively evaluating uplift methods. +""" + +from __future__ import annotations + +from benchmarking.synthetic.cp_core import HOT_CP_MODEL, CpCore, CpParams, cp_surface +from benchmarking.synthetic.generator import SyntheticDataset, ToggleSchedule, generate_dataset, treated_mask +from benchmarking.synthetic.ground_truth import UpliftResult, true_uplift +from benchmarking.synthetic.plots import plot_power_curve_comparison +from benchmarking.synthetic.schema import ColumnSchema +from benchmarking.synthetic.sources.hill_of_towie import ( + HOT_ACTIVE_POWER_STAT_COLS, + HOT_COLUMNS, + HOT_HUB_HEIGHT_M, + HOT_RATED_POWER_KW, +) +from benchmarking.synthetic.upgrades import ( + ConditionCpChange, + ConstantCpChange, + RatedPowerChange, + UpgradeEffect, + WindSpeedCpChange, + apply_upgrades, +) + +__all__ = [ + "HOT_ACTIVE_POWER_STAT_COLS", + "HOT_COLUMNS", + "HOT_CP_MODEL", + "HOT_HUB_HEIGHT_M", + "HOT_RATED_POWER_KW", + "ColumnSchema", + "ConditionCpChange", + "ConstantCpChange", + "CpCore", + "CpParams", + "RatedPowerChange", + "SyntheticDataset", + "ToggleSchedule", + "UpgradeEffect", + "UpliftResult", + "WindSpeedCpChange", + "apply_upgrades", + "cp_surface", + "generate_dataset", + "plot_power_curve_comparison", + "treated_mask", + "true_uplift", +] diff --git a/benchmarking/synthetic/cp_core.py b/benchmarking/synthetic/cp_core.py new file mode 100644 index 00000000..9af62805 --- /dev/null +++ b/benchmarking/synthetic/cp_core.py @@ -0,0 +1,192 @@ +"""Vectorised Cp-space physics core for synthetic upgrade injection. + +The Cp surface is an analytic, parameterised model of Cp as a function of tip-speed +ratio (TSR) and blade pitch. It makes a made-up surface that is internally self-consistent for power<->Cp +conversion. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +import numpy as np +import numpy.typing as npt + + +@dataclass(frozen=True) +class CpParams: + """Parameters of the analytic ``Cp(TSR, pitch)`` surface.""" + + cp_max: float + opt_pitch: float + pitch_scale: float + opt_tsr: float + tsr_scale: float + banana_factor: float + + +# Hill of Towie Cp model: log-symmetric TSR form, coefficients least-squares fitted to +# the turbine's Cp(TSR, pitch) table, weighted (1 + Cp)^2 to favour the high-Cp region. +HOT_CP_MODEL = CpParams( + cp_max=0.4509, + opt_pitch=-1.2085, + pitch_scale=0.0067, + opt_tsr=6.6255, + tsr_scale=1.5558, + banana_factor=42.2606, +) + + +def cp_surface( + *, + tsr: npt.ArrayLike, + pitch: npt.ArrayLike, + params: CpParams = HOT_CP_MODEL, +) -> npt.NDArray[np.float64]: + """Evaluate the analytic Cp surface at the given TSR and pitch. + + :param tsr: tip-speed ratio (scalar or array) + :param pitch: blade pitch in degrees (scalar or array) + :param params: Cp surface parameters + :return: Cp value(s) + """ + tsr_arr = np.asarray(tsr, dtype=float) + pitch_arr = np.asarray(pitch, dtype=float) + + # banana_factor shifts the effective optimal pitch as TSR departs from optimal, + # producing the characteristic curved ("banana") Cp contours. + effective_opt_pitch = params.opt_pitch + params.banana_factor * np.abs(1.0 / params.opt_tsr - 1.0 / tsr_arr) + pitch_factor = np.maximum(0.0, 1.0 - params.pitch_scale * (effective_opt_pitch - pitch_arr) ** 2) + # The TSR falloff is log-symmetric (penalises tsr/opt_tsr + opt_tsr/tsr - 2), so it + # decays gently up the high-TSR side as real turbines do, unlike a symmetric + # quadratic. + tsr_ratio = tsr_arr / params.opt_tsr + tsr_factor = np.maximum(0.0, 1.0 - params.tsr_scale * (tsr_ratio + 1.0 / tsr_ratio - 2.0)) + return params.cp_max * pitch_factor * tsr_factor + + +# Defaults for the power-based region-2 fraction, ported from baby-yoda's HoT turbine +# model (sigmoid midpoint 2000 kW, steepness 1/130, for a 2300 kW rated turbine). +REGION2_MIDPOINT_KW = 2000.0 +REGION2_STEEPNESS = 1.0 / 130.0 + +# A Cp change is not applied to records at or above this fraction of rated power: such +# points are effectively at pure rated and should be left unchanged (the region-2 sigmoid +# only tails towards zero there, so without this guard a tiny change would still leak in). +NEAR_RATED_POWER_FRACTION = 0.995 + + +def region2_fraction( + power_kw: npt.ArrayLike, + *, + midpoint_kw: float = REGION2_MIDPOINT_KW, + steepness: float = REGION2_STEEPNESS, +) -> npt.NDArray[np.float64]: + """Estimate the fraction of a 10-min period spent in region 2 from mean power. + + A squared logistic that is ~1 deep in region 2 and tails to ~0 as power approaches + rated, so a Cp change applied in region 2 fades out near rated power. + + :param power_kw: mean active power (scalar or array) + :param midpoint_kw: power at which the underlying logistic is 0.5 + :param steepness: logistic steepness + :return: region-2 fraction in [0, 1] + """ + power_arr = np.asarray(power_kw, dtype=float) + logistic = 1.0 / (1.0 + np.exp(steepness * (power_arr - midpoint_kw))) + return logistic**2 + + +def power_from_cp_change( + baseline_power_kw: npt.ArrayLike, + *, + cp_ratio: npt.ArrayLike, + rated_power_kw: float, +) -> npt.NDArray[np.float64]: + """Apply a Cp ratio to baseline power, weighted by the region-2 fraction. + + Only the region-2 fraction of the period responds to the Cp change; the remainder + (near/at rated) is unchanged, and the result is clipped so it never exceeds the + larger of rated power and the original power. A Cp change is applied only to records + that are *producing and below pure rated*: non-producing records (zero or negative + power: idling, self-consumption, curtailment) and virtually-rated records (at or above + ``NEAR_RATED_POWER_FRACTION`` of rated) are left untouched. + + :param baseline_power_kw: original mean active power (scalar or array) + :param cp_ratio: ratio of new Cp to baseline Cp (e.g. 1.02 for +2%) + :param rated_power_kw: rated power used for the upper clip + :return: modified mean active power + """ + baseline = np.asarray(baseline_power_kw, dtype=float) + ratio = np.asarray(cp_ratio, dtype=float) + fraction = region2_fraction(baseline) + new_power = baseline * (1.0 + fraction * (ratio - 1.0)) + upper_clip = np.maximum(baseline, rated_power_kw) + clipped = np.minimum(new_power, upper_clip) + modifiable = (baseline > 0.0) & (baseline < rated_power_kw * NEAR_RATED_POWER_FRACTION) + return np.where(modifiable, clipped, baseline) + + +# Generator-speed vs power curve for the HoT turbine, ported from baby-yoda +# (operating curves, sco_hot). +_RPM_VS_POWER_X = [0.0, 200.0, 350.0, 500.0, 750.0, 1000.0, 1250.0, 1500.0, 1750.0, 2000.0, 2300.0] +_RPM_VS_POWER_Y = [720.0, 755.0, 956.0, 1090.0, 1262.0, 1392.0, 1471.0, 1512.0, 1534.0, 1545.0, 1552.0] + + +def rpm_from_power(power_kw: npt.ArrayLike) -> npt.NDArray[np.float64]: + """Look up generator rpm from mean power on the ported operating curve.""" + return np.interp(np.asarray(power_kw, dtype=float), _RPM_VS_POWER_X, _RPM_VS_POWER_Y) + + +def rpm_from_power_change( + *, + baseline_rpm: npt.ArrayLike, + baseline_power_kw: npt.ArrayLike, + new_power_kw: npt.ArrayLike, +) -> npt.NDArray[np.float64]: + """Scale baseline rpm by the operating-curve rpm ratio implied by a power change. + + rpm tracks power along the operating curve, so a change in power drags rpm by the + ratio of curve rpm at the new vs baseline power. Where the new power is not + positive the baseline rpm is left unchanged. + + :param baseline_rpm: original generator rpm + :param baseline_power_kw: original mean active power + :param new_power_kw: modified mean active power + :return: modified generator rpm + """ + baseline_rpm_arr = np.asarray(baseline_rpm, dtype=float) + new_power = np.asarray(new_power_kw, dtype=float) + ratio = rpm_from_power(new_power) / rpm_from_power(baseline_power_kw) + scaled = baseline_rpm_arr * ratio + return np.where(new_power > 0.0, scaled, baseline_rpm_arr) + + +@dataclass(frozen=True) +class CpCore: + """Per-turbine Cp-space physics, the integration point upgrades operate through. + + Bundles the rated power and Cp surface parameters so an upgrade can turn a desired + Cp ratio into modified power and rpm without knowing turbine-specific constants. + """ + + rated_power_kw: float = 2300.0 + cp_params: CpParams = HOT_CP_MODEL + + def apply_cp_ratio(self, baseline_power_kw: npt.ArrayLike, *, cp_ratio: npt.ArrayLike) -> npt.NDArray[np.float64]: + """Apply a Cp ratio to baseline power using this core's rated-power clip.""" + return power_from_cp_change(baseline_power_kw, cp_ratio=cp_ratio, rated_power_kw=self.rated_power_kw) + + def rpm_after( + self, + *, + baseline_rpm: npt.ArrayLike, + baseline_power_kw: npt.ArrayLike, + new_power_kw: npt.ArrayLike, + ) -> npt.NDArray[np.float64]: + """Return generator rpm after a power change, tracking the operating curve.""" + return rpm_from_power_change( + baseline_rpm=baseline_rpm, + baseline_power_kw=baseline_power_kw, + new_power_kw=new_power_kw, + ) diff --git a/benchmarking/synthetic/generator.py b/benchmarking/synthetic/generator.py new file mode 100644 index 00000000..526538a7 --- /dev/null +++ b/benchmarking/synthetic/generator.py @@ -0,0 +1,195 @@ +"""Orchestration: turn real SCADA + an upgrade recipe into a synthetic dataset. + +A stable, no-upgrade window of real SCADA is taken as the baseline. The chosen test +turbine(s) have a known upgrade injected into their treated rows (post-changeover in +``prepost`` mode); references and untreated rows stay real. The untouched original is +retained alongside so the true uplift can always be derived by comparison. +""" + +from __future__ import annotations + +import json +from dataclasses import asdict, dataclass, field +from pathlib import Path +from typing import TYPE_CHECKING, Literal + +import numpy as np + +from benchmarking.synthetic.cp_core import HOT_CP_MODEL, CpCore, CpParams +from benchmarking.synthetic.ground_truth import UpliftResult, true_uplift +from benchmarking.synthetic.sources.hill_of_towie import HOT_COLUMNS +from benchmarking.synthetic.upgrades import apply_upgrades + +if TYPE_CHECKING: + import pandas as pd + + from benchmarking.synthetic.schema import ColumnSchema + +_GROUND_TRUTH_WS_BINS = list(np.arange(0.0, 26.0, 1.0)) + + +@dataclass(frozen=True) +class ToggleSchedule: + """A simple regular toggle: the upgrade alternates on/off over each ``period``. + + :param period: length of one full on/off cycle (half off, half on) + :param start_on: whether the first block is treated (on); default off (pre-like) + :param start: when toggling begins; rows before it are untreated baseline and it is + the toggle origin. ``None`` (default) keeps the data's first timestamp as origin + with no baseline. + """ + + period: pd.Timedelta + start_on: bool = False + start: pd.Timestamp | None = None + + +@dataclass +class SyntheticDataset: + """A generated synthetic dataset plus its untouched ground-truth reference.""" + + synthetic_df: pd.DataFrame + original_df: pd.DataFrame + run_metadata: dict = field(default_factory=dict) + columns: ColumnSchema = HOT_COLUMNS + + def true_uplift( + self, + *, + test_wtg: str | None = None, + mask: np.ndarray | None = None, + by: str | None = None, + bins: list | None = None, + ) -> UpliftResult: + """Derive the true uplift by comparing synthetic to original. + + Defaults ``test_wtg`` to the first test turbine in the run metadata. + """ + if test_wtg is None: + test_wtg = self.run_metadata["test_wtgs"][0] + return true_uplift( + self.synthetic_df, self.original_df, test_wtg=test_wtg, mask=mask, by=by, bins=bins, columns=self.columns + ) + + def save(self, out_dir: str | Path) -> Path: + """Write synthetic.parquet, original.parquet and run_metadata.json to ``out_dir``. + + The metadata file includes a full-record ground-truth summary (overall uplift and + a per-original-wind-speed-bin breakdown) for each test turbine. + """ + out_path = Path(out_dir) + out_path.mkdir(parents=True, exist_ok=True) + self.synthetic_df.to_parquet(out_path / "synthetic.parquet") + self.original_df.to_parquet(out_path / "original.parquet") + + ground_truth = {} + for wtg in self.run_metadata.get("test_wtgs", []): + result = self.true_uplift(test_wtg=wtg, by="ws", bins=_GROUND_TRUTH_WS_BINS) + assert result.by_condition is not None # always set when ``by`` is given # noqa: S101 + by_condition = result.by_condition.copy() + by_condition["condition_bin"] = by_condition["condition_bin"].astype(str) + ground_truth[wtg] = { + "overall": result.overall, + "by_wind_speed": by_condition.to_dict(orient="records"), + } + + metadata = {**self.run_metadata, "ground_truth": ground_truth} + (out_path / "run_metadata.json").write_text(json.dumps(metadata, indent=2, default=str)) + return out_path + + +def _treated_mask( + index: pd.DatetimeIndex, + *, + mode: Literal["prepost", "toggle"], + upgrade_timing: pd.Timestamp | ToggleSchedule, +) -> np.ndarray: + """Boolean mask over ``index`` selecting the rows where the upgrade is active.""" + if mode == "prepost": + if isinstance(upgrade_timing, ToggleSchedule): + msg = "prepost mode needs a changeover Timestamp, got a ToggleSchedule" + raise TypeError(msg) + return np.asarray(index >= upgrade_timing) + if mode == "toggle": + if not isinstance(upgrade_timing, ToggleSchedule): + msg = f"toggle mode needs a ToggleSchedule, got {type(upgrade_timing).__name__}" + raise TypeError(msg) + schedule = upgrade_timing + origin = schedule.start if schedule.start is not None else index.min() + # ``period`` is a full on/off cycle, so each on/off block is half a period. + block = (index - origin) // (schedule.period / 2) + on_parity = 0 if schedule.start_on else 1 + treated = np.asarray((np.asarray(block) % 2) == on_parity) + if schedule.start is not None: + # Rows before toggling begins are untreated baseline (and their negative block + # index must not be allowed to alias onto an "on" parity). + treated &= np.asarray(index >= schedule.start) + return treated + msg = f"unknown mode {mode!r}" + raise ValueError(msg) + + +def treated_mask(index: pd.DatetimeIndex, upgrade_timing: pd.Timestamp | ToggleSchedule) -> np.ndarray: + """Boolean mask of the rows an upgrade treats, with the mode inferred from ``upgrade_timing``. + + A ``ToggleSchedule`` selects toggle mode; any other value (a changeover timestamp) selects + prepost mode. Shared by the generator, methods and the scoring truth path so they all agree + on which rows are treated. + """ + mode: Literal["prepost", "toggle"] = "toggle" if isinstance(upgrade_timing, ToggleSchedule) else "prepost" + return _treated_mask(index, mode=mode, upgrade_timing=upgrade_timing) + + +def generate_dataset( + *, + scada_df: pd.DataFrame, + test_wtgs: list[str], + upgrades: list, + mode: Literal["prepost", "toggle"], + upgrade_timing: pd.Timestamp | ToggleSchedule, + cp_params: CpParams = HOT_CP_MODEL, + rated_power_kw: float = 2300.0, + columns: ColumnSchema = HOT_COLUMNS, + seed: int = 0, +) -> SyntheticDataset: + """Generate a synthetic dataset by injecting an upgrade into the test turbine(s). + + :param scada_df: source-native long real SCADA (all turbines), the no-upgrade baseline + :param test_wtgs: turbine name(s) to upgrade + :param upgrades: upgrade callables applied to each test turbine's treated rows + :param mode: ``"prepost"`` (changeover date) or ``"toggle"`` + :param upgrade_timing: changeover timestamp (prepost) or toggle schedule + :param cp_params: Cp surface parameters for the test turbines + :param rated_power_kw: baseline rated power for the test turbines + :param columns: the source-native column schema ``scada_df`` is keyed by + :param seed: recorded in run metadata for provenance; generation is fully + deterministic and does not otherwise consume it + :return: the synthetic dataset, original reference and run metadata + """ + original_df = scada_df.copy() + synthetic_df = scada_df.copy() + modified_columns = (columns.active_power, columns.gen_rpm, columns.wind_speed) + + treated = _treated_mask(synthetic_df.index, mode=mode, upgrade_timing=upgrade_timing) + for wtg in test_wtgs: + is_test = (synthetic_df[columns.turbine] == wtg).to_numpy() + mask = is_test & treated + if not mask.any(): + continue + cp = CpCore(rated_power_kw=rated_power_kw, cp_params=cp_params) + modified = apply_upgrades(synthetic_df.loc[mask], upgrades, cp=cp, columns=columns) + for col in modified_columns: + synthetic_df.loc[mask, col] = modified[col].to_numpy() + + run_metadata = { + "test_wtgs": list(test_wtgs), + "mode": mode, + "upgrade_timing": str(upgrade_timing), + "upgrades": [u.description for u in upgrades], + "rated_power_kw": rated_power_kw, + "cp_params": asdict(cp_params), + "seed": seed, + } + return SyntheticDataset( + synthetic_df=synthetic_df, original_df=original_df, run_metadata=run_metadata, columns=columns + ) diff --git a/benchmarking/synthetic/ground_truth.py b/benchmarking/synthetic/ground_truth.py new file mode 100644 index 00000000..17acf828 --- /dev/null +++ b/benchmarking/synthetic/ground_truth.py @@ -0,0 +1,133 @@ +"""Comparison-derived ground-truth uplift for synthetic datasets. + +The true uplift is never a declared constant: it is computed by comparing the test +turbine's synthetic power to its original power over exactly the records used, so it +changes with campaign length and any condition filter. Conditions for per-condition +breakdowns use the original (treatment-invariant) signals. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd + +from benchmarking.synthetic.sources.hill_of_towie import HOT_COLUMNS + +if TYPE_CHECKING: + import numpy.typing as npt + + from benchmarking.synthetic.schema import ColumnSchema + + +def _condition_series(rows: pd.DataFrame, by: str, columns: ColumnSchema) -> npt.NDArray[np.float64]: + """Resolve a treatment-invariant condition signal from the original rows.""" + if by == "ti": + ws = rows[columns.wind_speed].to_numpy(dtype=float) + sd = rows[columns.wind_speed_sd].to_numpy(dtype=float) + # NaN (not inf/0-division warning) for calm rows; warnings are errors in tests. + return np.divide(sd, ws, out=np.full_like(sd, np.nan), where=ws != 0) + if by == "ws": + return rows[columns.wind_speed].to_numpy(dtype=float) + if by == "power": + # The baseline operating point: bin on the ORIGINAL (pre-upgrade) active power so the axis is + # upgrade-invariant. + return rows[columns.active_power].to_numpy(dtype=float) + return rows[by].to_numpy(dtype=float) + + +@dataclass +class UpliftResult: + """Ground-truth uplift over a set of records.""" + + overall: float + by_condition: pd.DataFrame | None = None + + +def true_uplift( + synthetic_df: pd.DataFrame, + original_df: pd.DataFrame, + *, + test_wtg: str, + mask: npt.ArrayLike | None = None, + by: str | None = None, + bins: npt.ArrayLike | None = None, + columns: ColumnSchema = HOT_COLUMNS, +) -> UpliftResult: + """Compute the true uplift of the test turbine, synthetic vs original. + + :param synthetic_df: method-facing synthetic SCADA + :param original_df: untouched original SCADA (ground-truth reference) + :param test_wtg: the upgraded turbine to measure + :param mask: boolean selection over the test turbine's rows (time order); default is + the records the upgrade actually changed + :param by: optional treatment-invariant condition for a per-condition breakdown + (``"ws"``, ``"ti"`` or an original column name) + :param bins: bin edges for the ``by`` condition + :param columns: the source-native column schema the frames are keyed by + :return: the overall energy-ratio uplift, and a per-condition table when ``by`` is set + """ + if by is not None and bins is None: + msg = "bins must be provided when by is set (per-condition breakdown needs explicit edges)" + raise ValueError(msg) + + original_wtg = original_df[original_df[columns.turbine] == test_wtg] + synthetic_power = synthetic_df.loc[synthetic_df[columns.turbine] == test_wtg, columns.active_power].to_numpy( + dtype=float + ) + original_power = original_wtg[columns.active_power].to_numpy(dtype=float) + + row_mask = changed_record_mask(synthetic_power, original_power) if mask is None else np.asarray(mask, dtype=bool) + # Real SCADA carries NaN power (downtime/missing); such records have no usable energy + # and must not poison the sums, so the ratio is taken over finite records only. + effective = row_mask & np.isfinite(synthetic_power) & np.isfinite(original_power) + + denom = original_power[effective].sum() + overall = synthetic_power[effective].sum() / denom - 1.0 if denom else float("nan") + + by_condition = None + if by is not None: + by_condition = _uplift_by_condition( + condition=_condition_series(original_wtg, by, columns)[effective], + synthetic_power=synthetic_power[effective], + original_power=original_power[effective], + bins=bins, + ) + return UpliftResult(overall=float(overall), by_condition=by_condition) + + +def changed_record_mask( + synthetic_power: npt.NDArray[np.float64], original_power: npt.NDArray[np.float64] +) -> npt.NDArray[np.bool_]: + """Boolean mask of records the upgrade actually changed (NaN-safe). + + A plain ``synthetic != original`` would flag downtime rows where both powers are NaN + (since ``NaN != NaN``); those are excluded here so only genuinely modified records + are treated as upgraded. + """ + differs = synthetic_power != original_power + both_nan = np.isnan(synthetic_power) & np.isnan(original_power) + return differs & ~both_nan + + +def _uplift_by_condition( + *, + condition: npt.NDArray[np.float64], + synthetic_power: npt.NDArray[np.float64], + original_power: npt.NDArray[np.float64], + bins: npt.ArrayLike | None, +) -> pd.DataFrame: + """Energy-ratio uplift within bins of a condition signal.""" + grouped = pd.DataFrame( + { + "condition_bin": pd.cut(condition, bins=bins), + "synthetic_energy": synthetic_power, + "original_energy": original_power, + } + ).groupby("condition_bin", observed=False) + table = grouped[["synthetic_energy", "original_energy"]].sum() + table["n_records"] = grouped.size() + table["true_uplift"] = table["synthetic_energy"] / table["original_energy"] - 1.0 + return table.reset_index() diff --git a/benchmarking/synthetic/make_example_datasets.py b/benchmarking/synthetic/make_example_datasets.py new file mode 100644 index 00000000..50de0315 --- /dev/null +++ b/benchmarking/synthetic/make_example_datasets.py @@ -0,0 +1,173 @@ +"""Driver: produce one synthetic dataset per Issue 1 profile. + +``generate_example_datasets`` is source-agnostic (give it any wind-up-format SCADA). +``main`` wires it to the Hill of Towie open data so the whole thing runs end-to-end. +""" + +from __future__ import annotations + +import logging +import os +from pathlib import Path +from typing import Literal + +import pandas as pd + +from benchmarking.synthetic.generator import SyntheticDataset, ToggleSchedule, generate_dataset +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada +from benchmarking.synthetic.upgrades import ( + ConditionCpChange, + ConstantCpChange, + RatedPowerChange, + WindSpeedCpChange, +) + +logger = logging.getLogger(__name__) + +# A stable, no-upgrade Hill of Towie window: comfortably before the real T13 AeroUp +# (installed Sep 2021) +# All of 2016-2020 was previously confirmed stable for T01 and all nearby turbines +# during the setup of the Kaggle power prediction challenge. +DEFAULT_START_DT = pd.Timestamp("2020-06-01", tz="UTC") +DEFAULT_END_DT_EXCL = pd.Timestamp("2020-09-01", tz="UTC") +DEFAULT_TEST_WTG = "T01" + + +def default_output_root() -> Path: + """Return the root directory example datasets are written under. + + Overridable via the ``WIND_UP_BENCHMARKING_OUTPUT_DIR`` environment variable; + defaults to ``~/temp/wind-up-benchmarking/synthetic``. + """ + return Path( + os.getenv("WIND_UP_BENCHMARKING_OUTPUT_DIR", Path.home() / "temp" / "wind-up-benchmarking" / "synthetic") + ) + + +def example_profiles() -> dict[str, list]: + """Return the four Issue 1 upgrade profiles with concrete example parameters. + + - ``constant_cp``: a flat +3% region-2 Cp change (e.g. a blade add-on). + - ``wind_speed_cp``: a region-2 Cp change that peaks mid-region and tails to 0 at + rated (the AeroUp shape). + - ``ti_cp``: a Cp change that is larger at low turbulence intensity. + - ``rated_power``: a +5% rated-power uprate. + """ + return { + "constant_cp": [ConstantCpChange(delta=0.03)], + "wind_speed_cp": [WindSpeedCpChange(ws_points=(4.0, 7.0, 10.0, 13.0), deltas=(0.0, 0.04, 0.02, 0.0))], + "ti_cp": [ConditionCpChange(by="ti", points=(0.05, 0.15), deltas=(0.04, 0.0))], + "rated_power": [RatedPowerChange(new_rated_power_kw=2415.0)], + } + + +def generate_example_datasets( + *, + scada_df: pd.DataFrame, + test_wtgs: list[str], + mode: Literal["prepost", "toggle"], + upgrade_timing: pd.Timestamp | ToggleSchedule, + out_root: str | Path | None = None, + seed: int = 0, + save_plots: bool = True, +) -> dict[str, SyntheticDataset]: + """Generate (and optionally save) one synthetic dataset per example profile. + + :param scada_df: wind-up-format real SCADA (all turbines), the no-upgrade baseline + :param test_wtgs: turbine name(s) to upgrade + :param mode: ``"prepost"`` or ``"toggle"`` + :param upgrade_timing: changeover timestamp (prepost) or toggle schedule + :param out_root: if given, each dataset is saved under ``out_root / `` + :param seed: top-level seed for reproducibility + :param save_plots: when saving, also write a power-curve comparison PNG per test turbine + :return: mapping of profile name to its SyntheticDataset + """ + datasets: dict[str, SyntheticDataset] = {} + for name, upgrades in example_profiles().items(): + dataset = generate_dataset( + scada_df=scada_df, + test_wtgs=test_wtgs, + upgrades=upgrades, + mode=mode, + upgrade_timing=upgrade_timing, + seed=seed, + ) + if out_root is not None: + dataset_dir = Path(out_root) / name + dataset.save(dataset_dir) + if save_plots: + _save_power_curve_plots(dataset, dataset_dir, profile=name) + datasets[name] = dataset + return datasets + + +def _save_power_curve_plots(dataset: SyntheticDataset, dataset_dir: Path, *, profile: str) -> None: + """Write an original-vs-synthetic power-curve PNG for each test turbine.""" + # Imported lazily so the driver imports without matplotlib when plots aren't wanted. + import matplotlib.pyplot as plt # noqa: PLC0415 + + from benchmarking.synthetic.plots import plot_power_curve_comparison # noqa: PLC0415 + + for wtg in dataset.run_metadata.get("test_wtgs", []): + uplift = dataset.true_uplift(test_wtg=wtg).overall + fig = plot_power_curve_comparison( + dataset.synthetic_df, + dataset.original_df, + test_wtg=wtg, + save_path=dataset_dir / f"power_curve_{wtg}.png", + title=f"{profile}: {wtg} power curve (true uplift {uplift:+.2%})", + ) + plt.close(fig) + + +def main( + *, + out_root: str | Path | None = None, + data_dir: str | Path | None = None, + start_dt: pd.Timestamp = DEFAULT_START_DT, + end_dt_excl: pd.Timestamp = DEFAULT_END_DT_EXCL, + test_wtg: str = DEFAULT_TEST_WTG, + seed: int = 0, +) -> dict[str, SyntheticDataset]: + """Produce one synthetic dataset per Issue 1 profile from real Hill of Towie data. + + Downloads (and caches) the open Hill of Towie SCADA for a stable, no-upgrade window, + injects each example upgrade into ``test_wtg`` as a ``prepost`` changeover at the + middle of the window, and saves the datasets under ``out_root``. + + :param out_root: dataset output root; defaults to :func:`default_output_root` + :param data_dir: Hill of Towie data/cache dir; defaults to the package default + :param start_dt: inclusive UTC window start + :param end_dt_excl: exclusive UTC window end + :param test_wtg: turbine to upgrade + :param seed: top-level seed for reproducibility + :return: mapping of profile name to its SyntheticDataset + """ + out_root = Path(out_root) if out_root is not None else default_output_root() + scada_df, _metadata_df = load_hot_scada( + start_dt=start_dt, + end_dt_excl=end_dt_excl, + data_dir=Path(data_dir) if data_dir is not None else None, + ) + upgrade_timing = start_dt + (end_dt_excl - start_dt) / 2 + logger.info( + "Generating example datasets for %s over %s..%s (changeover %s) into %s", + test_wtg, + start_dt, + end_dt_excl, + upgrade_timing, + out_root, + ) + return generate_example_datasets( + scada_df=scada_df, + test_wtgs=[test_wtg], + mode="prepost", + upgrade_timing=upgrade_timing, + out_root=out_root, + seed=seed, + ) + + +if __name__ == "__main__": + logging.basicConfig(level=logging.INFO) + main() diff --git a/benchmarking/synthetic/plots.py b/benchmarking/synthetic/plots.py new file mode 100644 index 00000000..c0b896ba --- /dev/null +++ b/benchmarking/synthetic/plots.py @@ -0,0 +1,100 @@ +"""Verification plots for synthetic datasets. + +Three panels for the test turbine, all sharing the wind-speed x-axis: + +1. *original* power curve (power vs wind speed); +2. *synthetic* power curve, on the same power y-axis as the original so the injected + upgrade is directly comparable; +3. the per-record **kW change** (synthetic minus original) vs wind speed for the + treated records, which makes the injected uplift shape easy to read. + +Treated (post-upgrade) records are highlighted in the first two panels so you can +confirm the injection lands where expected and leaves the baseline rows untouched. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np + +from benchmarking.synthetic.ground_truth import changed_record_mask +from benchmarking.synthetic.sources.hill_of_towie import HOT_COLUMNS + +if TYPE_CHECKING: + from pathlib import Path + + import pandas as pd + from matplotlib.figure import Figure + + from benchmarking.synthetic.schema import ColumnSchema + +_BASELINE_STYLE = {"s": 6, "alpha": 0.4, "color": "tab:blue", "label": "baseline rows"} +_TREATED_STYLE = {"s": 6, "alpha": 0.5, "color": "tab:red", "label": "treated rows"} + + +def plot_power_curve_comparison( + synthetic_df: pd.DataFrame, + original_df: pd.DataFrame, + *, + test_wtg: str, + save_path: str | Path | None = None, + title: str | None = None, + columns: ColumnSchema = HOT_COLUMNS, +) -> Figure: + """Plot the test turbine's original vs synthetic power curve plus the kW change. + + The original and synthetic power-curve panels share x (wind speed) and y (power) + limits and gridlines; a third panel shows the synthetic-minus-original power change + against wind speed for the treated records. Records the upgrade actually changed + (NaN-safe) are highlighted in the first two panels. + + :param synthetic_df: source-native synthetic SCADA (all turbines), keyed by ``columns`` + :param original_df: the untouched source-native original SCADA (all turbines) + :param test_wtg: turbine to plot + :param save_path: if given, the figure is written here (PNG) + :param title: optional overall figure title + :param columns: the source-native column schema the frames are keyed by + :return: the matplotlib Figure + """ + original = original_df[original_df[columns.turbine] == test_wtg] + synthetic = synthetic_df[synthetic_df[columns.turbine] == test_wtg] + + ws = original[columns.wind_speed].to_numpy(dtype=float) + original_power = original[columns.active_power].to_numpy(dtype=float) + synthetic_power = synthetic[columns.active_power].to_numpy(dtype=float) + + # Treated = records genuinely modified by the upgrade (NaN downtime rows excluded). + treated = changed_record_mask(synthetic_power, original_power) + + fig, (ax_orig, ax_syn, ax_delta) = plt.subplots(1, 3, figsize=(17, 5), sharex=True) + ax_syn.sharey(ax_orig) # tie the two power-curve y-axes; the kW-change panel is its own + + for ax, power, panel_title in ( + (ax_orig, original_power, "Original"), + (ax_syn, synthetic_power, "Synthetic"), + ): + ax.scatter(ws[~treated], power[~treated], **_BASELINE_STYLE) + ax.scatter(ws[treated], power[treated], **_TREATED_STYLE) + ax.set_title(panel_title) + ax.set_xlabel("Wind speed [m/s]") + ax.grid(visible=True, alpha=0.3) + ax.legend(loc="lower right", markerscale=2) + ax_orig.set_ylabel("Active power [kW]") + + delta = synthetic_power - original_power + finite_treated = treated & np.isfinite(delta) + ax_delta.scatter(ws[finite_treated], delta[finite_treated], s=6, alpha=0.5, color="tab:red") + ax_delta.axhline(0.0, color="k", linewidth=0.8) + ax_delta.set_title("Injected change (synthetic - original)") + ax_delta.set_xlabel("Wind speed [m/s]") + ax_delta.set_ylabel("Power change [kW]") + ax_delta.grid(visible=True, alpha=0.3) + + fig.suptitle(title if title is not None else f"{test_wtg} power curve: original vs synthetic") + fig.tight_layout() + + if save_path is not None: + fig.savefig(save_path, dpi=150) + return fig diff --git a/benchmarking/synthetic/schema.py b/benchmarking/synthetic/schema.py new file mode 100644 index 00000000..e7529644 --- /dev/null +++ b/benchmarking/synthetic/schema.py @@ -0,0 +1,82 @@ +"""The column vocabulary the synthetic pipeline and harness speak. + +The benchmarking layer is deliberately independent of v0: the synthetic generator, the +ground-truth comparison and the harness all operate on **source-native** SCADA column names +(the real tag names a data source ships), never on v0's :class:`~wind_up.constants.DataColumns` +aliases. A :class:`ColumnSchema` names the handful of semantic roles those components need, so +the only place that knows a source's actual column names is the source adapter, which provides +its own :class:`ColumnSchema` (e.g. ``HOT_COLUMNS`` for Hill of Towie). + +v0-specific aliasing is therefore not a pipeline concern at all: it lives entirely inside the +v0 baseline (which converts the source-native frame to wind-up format on the way in). +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING + +if TYPE_CHECKING: + from collections.abc import Iterable + + +@dataclass(frozen=True) +class ColumnSchema: + """The source-native column names for the semantic roles the pipeline reads. + + :param turbine: the turbine-identifier column of the long-format SCADA frame + :param active_power: mean active power + :param wind_speed: mean nacelle wind speed + :param wind_speed_sd: nacelle wind-speed standard deviation (turbulence intensity input) + :param gen_rpm: mean generator rpm + :param availability: a "ready to operate" counter (e.g. seconds available in the period). + **Required**: the methods use it for downtime filtering, which must never be silently + skipped, so every source must supply it. + :param active_power_min: the per-reference active-power **minimum** companion column (Issue 11 / + F12). Optional on the schema, but ``PowerModelMethod`` **requires** it — each reference's + power minimum is a standard model feature, not a per-driver opt-in — and validates its + presence on construction. Sources without it can still drive the lighter methods. + + The remaining roles are **diagnostics-only**: they name extra signals the shared per-run + diagnostics plot, never estimation inputs. Each defaults to ``None`` so a source that lacks a + signal (or a caller that does not care) leaves it unset and the corresponding plots skip + gracefully. ``nacelle_position`` in particular is a wind-direction *proxy* for plotting and + must **not** become a model feature — it is post-treatment / not treatment-invariant + (design-note §3). + + :param pitch: blade pitch angle (a representative single sensor is fine) + :param reactive_power: mean reactive power + :param nacelle_position: nacelle/yaw position (wind-direction proxy; diagnostics only) + :param ambient_temp: ambient temperature + :param exclude_row: names a caller-supplied **boolean** column marking rows to drop from row + selection (e.g. special operating modes the treatment cannot affect). Optional and off by + default; when set, a method that honours it excludes the *test* turbine's flagged rows + alongside the downtime filter. Must be all-``False`` where unknown (never NaN) so an expanded + time index never turns a gap into an exclusion; a method honouring the role raises on NaN. + """ + + turbine: str + active_power: str + wind_speed: str + wind_speed_sd: str + gen_rpm: str + availability: str + active_power_min: str | None = None + pitch: str | None = None + reactive_power: str | None = None + nacelle_position: str | None = None + ambient_temp: str | None = None + exclude_row: str | None = None + + def require_roles(self, roles: Iterable[str]) -> None: + """Raise ``ValueError`` if any named role is unset or blank (``None``, empty, or whitespace). + + The estimation methods take their column names *only* from a ``ColumnSchema`` (rather than + separate per-column arguments that could disagree with it), so each validates on construction + that the schema actually names every role it reads. ``roles`` are field names — an unknown one + is a programming error and raises ``AttributeError`` (the call sites pass literal role names). + """ + missing = [role for role in roles if not (getattr(self, role) or "").strip()] + if missing: + msg = f"columns is missing required role(s) {missing}: {self}" + raise ValueError(msg) diff --git a/benchmarking/synthetic/sources/__init__.py b/benchmarking/synthetic/sources/__init__.py new file mode 100644 index 00000000..148322ec --- /dev/null +++ b/benchmarking/synthetic/sources/__init__.py @@ -0,0 +1 @@ +"""Data-source adapters that load real SCADA for synthetic dataset generation.""" diff --git a/benchmarking/synthetic/sources/hill_of_towie.py b/benchmarking/synthetic/sources/hill_of_towie.py new file mode 100644 index 00000000..4d06a12a --- /dev/null +++ b/benchmarking/synthetic/sources/hill_of_towie.py @@ -0,0 +1,740 @@ +"""Hill of Towie open-data source adapter for the synthetic generator. + +A self-contained (vendored) copy of the pieces of the +``hill-of-towie-open-source-analysis`` ``hot_open`` package needed to load +wind-up-format SCADA end to end: + +- the Zenodo fetcher (``ensure_hot_data_files`` / ``download_zenodo_data``) that + downloads and caches the Hill of Towie v2 datapack (Zenodo record ``20204946``); +- the 10-minute SCADA loader (``load_hot_10min_data``) and the wide-to-long reshape + (``scada_wide_to_long``) that keeps source-native ``wtc_*`` tag names; +- ``load_hot_scada`` that ties them together and returns source-native long SCADA plus + turbine metadata; +- ``long_to_wind_up_format``, the v0-only on-ramp that aliases the source columns to + :class:`~wind_up.constants.DataColumns` names and derives ``PitchAngleMean`` / + ``ShutdownDuration``. + +Copied rather than imported so ``benchmarking`` stays hermetic and depends on +``wind_up`` only for :class:`~wind_up.constants.DataColumns` (used by the v0 on-ramp). +``requests`` and ``tqdm`` are imported lazily inside the network/IO functions so the pure +transforms import without them. +""" + +from __future__ import annotations + +import hashlib +import json +import logging +import math +import os +import time +from pathlib import Path +from typing import TYPE_CHECKING, NamedTuple +from zipfile import ZipFile + +import pandas as pd + +from benchmarking.synthetic.schema import ColumnSchema +from wind_up.constants import DataColumns + +if TYPE_CHECKING: + from collections.abc import Collection, Sequence + + import requests + +logger = logging.getLogger(__name__) + +TIMEBASE_S = 600 +HOT_V2_RECORD_ID = "20204946" +HOT_FIRST_WTG = 1 +HOT_LAST_WTG = 21 +_HOT_SERIAL_OFFSET = 2304509 + +BYTES_IN_1MB = 1024 * 1024 +CHUNK_SIZE = 10 * BYTES_IN_1MB +SMALL_FILE_THRESHOLD_BYTES = 2 * BYTES_IN_1MB + +# Network resilience knobs for streamed Zenodo downloads. ``timeout`` is passed to +# ``requests.get`` as a ``(connect, read)`` tuple; with ``stream=True`` the read +# value is the budget *between* received chunks. +_CONNECT_TIMEOUT_S = 10 +_READ_TIMEOUT_S = 60 +_MAX_DOWNLOAD_ATTEMPTS = 5 +_BACKOFF_BASE_S = 2.0 +_HTTP_PARTIAL_CONTENT = 206 + + +def get_data_dir() -> Path: + """Return the local Hill of Towie data/cache directory, creating it if needed. + + Overridable via the ``WIND_UP_BENCHMARKING_DATA_DIR`` environment variable; + defaults to ``~/temp/wind-up-benchmarking/data``. + """ + path = Path(os.getenv("WIND_UP_BENCHMARKING_DATA_DIR", Path.home() / "temp" / "wind-up-benchmarking" / "data")) + path.mkdir(parents=True, exist_ok=True) + return path + + +# -------------------------------------------------------------------------------------- +# Zenodo fetch +# -------------------------------------------------------------------------------------- +def download_zenodo_data( + record_id: str, + *, + output_dir: Path | None = None, + filenames: Collection[str] | None = None, + cache_overwrite: bool = False, +) -> None: + """Download and cache files from zenodo.org.""" + import requests # noqa: PLC0415 (lazy: keep network deps out of the import path) + + output_dir = output_dir if output_dir is not None else get_data_dir() + output_dir.mkdir(parents=True, exist_ok=True) + metadata_fpath = output_dir / "zenodo_dataset_metadata.json" + + # One Session for the whole download so its connection pool (and every socket) is + # closed deterministically on exit. A per-call ``requests.get`` closes its transient + # pool before the streamed response's socket is released back to it, leaking the + # socket until GC -- which trips ``filterwarnings = error`` via ResourceWarning. + with requests.Session() as session: + if not cache_overwrite and metadata_fpath.is_file(): + logger.info("Loading metadata from %s", metadata_fpath) + with metadata_fpath.open() as f: + content = json.load(f) + else: + logger.info("Fetching metadata from zenodo...") + with session.get( + f"https://zenodo.org/api/records/{record_id}", + timeout=(_CONNECT_TIMEOUT_S, _READ_TIMEOUT_S), + ) as r: + r.raise_for_status() + content = r.json() + with metadata_fpath.open("w") as f: + json.dump(content, f) + logger.info("Saved metadata to %s", metadata_fpath) + + remote_files: list[dict] = content["files"] + if filenames is None: + files_to_download: list[dict] = list(remote_files) + else: + files_to_download = list(_check_name_of_files_to_download(filenames, remote_files)) + required_keys = {f["key"] for f in files_to_download} + # Auto-include any small file in the record (READMEs, deployment reports, ...). + for rf in remote_files: + if rf["size"] < SMALL_FILE_THRESHOLD_BYTES and rf["key"] not in required_keys: + files_to_download.append(rf) + + downloaded_files = 0 + n_files_to_download = len(files_to_download) + for i_file, file_to_download in enumerate(files_to_download, start=1): + is_required = file_to_download["key"] in required_keys + downloaded_files += _download_one_file( + session, + file_to_download, + output_dir, + cache_overwrite=cache_overwrite, + is_required=is_required, + progress_prefix=f"[{i_file}/{n_files_to_download}]", + ) + logger.info("Download finished: %s new files cached at %s", downloaded_files, output_dir) + + +def _download_one_file( + session: requests.Session, + file_entry: dict, + output_dir: Path, + *, + cache_overwrite: bool, + is_required: bool, + progress_prefix: str, +) -> int: + """Download a single Zenodo file. Returns 1 if a new file was written, 0 otherwise. + + Uses the caller's :class:`requests.Session` so its connection pool is closed once + by the caller, releasing every socket deterministically rather than at GC. + + Retries up to ``_MAX_DOWNLOAD_ATTEMPTS`` times on transient network errors with + exponential backoff, resuming partial downloads via a ``Range`` header. Required + files re-raise after exhausting retries; optional small files warn and clean up. + """ + import requests # noqa: PLC0415 (lazy: keep network deps out of the import path) + from tqdm import tqdm # noqa: PLC0415 + + retryable: tuple[type[requests.RequestException], ...] = ( + requests.ConnectionError, + requests.Timeout, + requests.exceptions.ChunkedEncodingError, + ) + + _file_name = file_entry["key"] + _file_size = file_entry["size"] + _file_url = file_entry["links"]["self"] + dst_fpath = output_dir / _file_name + + if cache_overwrite and dst_fpath.is_file(): + dst_fpath.unlink() + + if dst_fpath.is_file() and dst_fpath.stat().st_size >= _file_size: + logger.info("%s File %s already exists. Skipping download.", progress_prefix, dst_fpath) + return 0 + + logger.info("%s Beginning file download from Zenodo: %s...", progress_prefix, _file_name) + for attempt in range(1, _MAX_DOWNLOAD_ATTEMPTS + 1): + is_last_attempt = attempt == _MAX_DOWNLOAD_ATTEMPTS + existing_size = dst_fpath.stat().st_size if dst_fpath.is_file() else 0 + headers = {"Range": f"bytes={existing_size}-"} if existing_size > 0 else {} + try: + result = session.get( + _file_url, + stream=True, + timeout=(_CONNECT_TIMEOUT_S, _READ_TIMEOUT_S), + headers=headers, + ) + result.raise_for_status() + # If we requested a Range but the server returned 200, it ignored it. + resume = existing_size > 0 and result.status_code == _HTTP_PARTIAL_CONTENT + if existing_size > 0 and not resume: + logger.info( + "%s Server did not honor Range request (status %s); restarting from byte 0.", + progress_prefix, + result.status_code, + ) + existing_size = 0 + file_mode = "ab" if resume else "wb" + remaining_bytes = max(0, _file_size - existing_size) + with ( + result, # close the streamed response (and its socket) deterministically, not at GC + Path.open(dst_fpath, file_mode) as f, + tqdm( + total=remaining_bytes, + unit="B", + unit_scale=True, + unit_divisor=1024, + desc=f"Downloading {_file_name} ({_file_size / BYTES_IN_1MB:.2f} MB)", + ) as pbar, + ): + for chunk in result.iter_content(chunk_size=CHUNK_SIZE): + f.write(chunk) + pbar.update(len(chunk)) + except retryable as e: + if not is_last_attempt: + partial_size = dst_fpath.stat().st_size if dst_fpath.is_file() else 0 + sleep_s = _BACKOFF_BASE_S * (2 ** (attempt - 1)) + logger.warning( + "%s Download attempt %d/%d for %s failed (%s). Have %d/%d bytes. Sleeping %.1fs before retrying.", + progress_prefix, + attempt, + _MAX_DOWNLOAD_ATTEMPTS, + _file_name, + e, + partial_size, + _file_size, + sleep_s, + ) + time.sleep(sleep_s) + continue + return _resolve_download_failure( + exc=e, + dst_fpath=dst_fpath, + file_name=_file_name, + is_required=is_required, + progress_prefix=progress_prefix, + ) + except requests.RequestException as e: + # Non-retryable (e.g. 4xx HTTPError). Resolve immediately. + return _resolve_download_failure( + exc=e, + dst_fpath=dst_fpath, + file_name=_file_name, + is_required=is_required, + progress_prefix=progress_prefix, + ) + else: + return 1 + + msg = "unreachable: retry loop should have returned or raised" + raise RuntimeError(msg) + + +def _resolve_download_failure( + *, + exc: requests.RequestException, + dst_fpath: Path, + file_name: str, + is_required: bool, + progress_prefix: str, +) -> int: + """Re-raise for required files; warn-and-clean for optional ones.""" + if is_required: + # Leave partial bytes on disk so a subsequent run can resume via Range. + raise exc + logger.warning( + "%s Failed to download optional small file %s: %s. Continuing.", + progress_prefix, + file_name, + exc, + ) + if dst_fpath.is_file(): + dst_fpath.unlink() + return 0 + + +def _missing_small_files_from_cached_metadata(target_dir: Path) -> list[str]: + """Return small-file keys absent from ``target_dir`` per cached Zenodo metadata. + + Empty list when the metadata cache is missing or unreadable; the next successful + fetch rewrites the cache. + """ + metadata_fpath = target_dir / "zenodo_dataset_metadata.json" + if not metadata_fpath.is_file(): + return [] + try: + with metadata_fpath.open() as f: + content = json.load(f) + except (OSError, json.JSONDecodeError): + return [] + return [ + rf["key"] + for rf in content.get("files", []) + if rf.get("size", math.inf) < SMALL_FILE_THRESHOLD_BYTES and not (target_dir / rf["key"]).is_file() + ] + + +def ensure_hot_data_files(filenames: Collection[str], *, data_dir: Path | None = None) -> None: + """Download missing Hill of Towie v2 data files from Zenodo. + + Idempotent: makes no network call when every requested file exists locally and + cached metadata shows no missing small files. + """ + target_dir = data_dir if data_dir is not None else get_data_dir() + requested = list(filenames) + missing_requested = [f for f in requested if not (target_dir / f).is_file()] + missing_small = _missing_small_files_from_cached_metadata(target_dir) + if not missing_requested and not missing_small: + logger.info( + "ensure_hot_data_files: all %d requested files already present at %s, skipping download", + len(requested), + target_dir, + ) + return + logger.info( + "ensure_hot_data_files: downloading from Zenodo record %s into %s " + "(missing requested: %s; missing small files: %s)", + HOT_V2_RECORD_ID, + target_dir, + missing_requested, + missing_small, + ) + download_zenodo_data(record_id=HOT_V2_RECORD_ID, output_dir=target_dir, filenames=missing_requested) + + +def _check_name_of_files_to_download(filenames: Collection[str], remote_files: Collection[dict]) -> Collection[dict]: + requested_filenames = set(filenames) + remote_filenames = {i["key"] for i in remote_files} + if not requested_filenames.issubset(remote_filenames): + msg = ( + "Could not find all files in the Zenodo record. " + f"Missing files: {requested_filenames.difference(remote_filenames)}" + ) + raise ValueError(msg) + return [i for i in remote_files if i["key"] in requested_filenames] + + +# -------------------------------------------------------------------------------------- +# 10-minute SCADA loading + wind-up-format conversion +# -------------------------------------------------------------------------------------- +class WPSBackupFileField(NamedTuple): + """Hill of Towie field and table mapping.""" + + alias: str + field_name: str + table_name: str + + +# Source-native Hill of Towie tag names referenced by ``HOT_COLUMNS`` (the source-native schema +# methods see). Defined once here and reused below in ``hill_of_towie_fields`` so each tag string +# lives in exactly one place, while ``HOT_COLUMNS`` is built directly from these tags rather than +# routed through the v0 ``DataColumns`` vocabulary (which stays confined to the on-ramp aliases). +_TAG_ACTIVE_POWER_MEAN = "wtc_ActPower_mean" +_TAG_WIND_SPEED_MEAN = "wtc_AcWindSp_mean" +_TAG_WIND_SPEED_SD = "wtc_AcWindSp_stddev" +_TAG_GEN_RPM_MEAN = "wtc_GenRpm_mean" +# Diagnostics-only tags (not estimation inputs): see ``ColumnSchema`` and the shared per-run +# diagnostics. ``wtc_NacelPos_mean`` is a wind-direction proxy for plotting only. +_TAG_PITCH_MEAN = "wtc_PitcPosA_mean" +_TAG_REACTIVE_POWER_MEAN = "wtc_ReactPwr_mean" +_TAG_NACELLE_POSITION_MEAN = "wtc_NacelPos_mean" +_TAG_AMBIENT_TEMP_MEAN = "wtc_AmbieTmp_mean" +_TAG_AVAILABILITY = "wtc_ScReToOp_timeon" +# Reference active-power companion statistics (Issue 11): within-period max/min/SD of active power. +# The SD in particular is a calibration-stable, farm-sited turbulence proxy a method may opt into +# as reference features; the mean stays the primary signal. +_TAG_ACTIVE_POWER_MAX = "wtc_ActPower_max" +_TAG_ACTIVE_POWER_MIN = "wtc_ActPower_min" +_TAG_ACTIVE_POWER_SD = "wtc_ActPower_stddev" + + +hill_of_towie_fields = [ + WPSBackupFileField( + alias=DataColumns.active_power_mean, field_name=_TAG_ACTIVE_POWER_MEAN, table_name="tblSCTurGrid" + ), + WPSBackupFileField(alias=DataColumns.active_power_sd, field_name=_TAG_ACTIVE_POWER_SD, table_name="tblSCTurGrid"), + WPSBackupFileField(alias="ActivePowerMax", field_name=_TAG_ACTIVE_POWER_MAX, table_name="tblSCTurGrid"), + WPSBackupFileField(alias="ActivePowerMin", field_name=_TAG_ACTIVE_POWER_MIN, table_name="tblSCTurGrid"), + WPSBackupFileField(alias="ReactivePowerMean", field_name="wtc_ReactPwr_mean", table_name="tblSCTurGrid"), + WPSBackupFileField(alias=DataColumns.wind_speed_mean, field_name=_TAG_WIND_SPEED_MEAN, table_name="tblSCTurbine"), + WPSBackupFileField(alias=DataColumns.wind_speed_sd, field_name=_TAG_WIND_SPEED_SD, table_name="tblSCTurbine"), + WPSBackupFileField(alias=DataColumns.yaw_angle_mean, field_name="wtc_NacelPos_mean", table_name="tblSCTurbine"), + WPSBackupFileField(alias=DataColumns.yaw_angle_min, field_name="wtc_NacelPos_min", table_name="tblSCTurbine"), + WPSBackupFileField(alias=DataColumns.yaw_angle_max, field_name="wtc_NacelPos_max", table_name="tblSCTurbine"), + WPSBackupFileField(alias=DataColumns.gen_rpm_mean, field_name=_TAG_GEN_RPM_MEAN, table_name="tblSCTurbine"), + WPSBackupFileField(alias="pitch_angle_a", field_name="wtc_PitcPosA_mean", table_name="tblSCTurbine"), + WPSBackupFileField(alias="pitch_angle_b", field_name="wtc_PitcPosB_mean", table_name="tblSCTurbine"), + WPSBackupFileField(alias="pitch_angle_c", field_name="wtc_PitcPosC_mean", table_name="tblSCTurbine"), + WPSBackupFileField(alias=DataColumns.ambient_temp, field_name="wtc_AmbieTmp_mean", table_name="tblSCTurTemp"), + WPSBackupFileField( + alias="Time ready to operate in period", field_name="wtc_ScReToOp_timeon", table_name="tblSCTurFlag" + ), + WPSBackupFileField(alias="YawOperationCounts", field_name="wtc_ScYawOpe_counts", table_name="tblSCTurFlag"), + WPSBackupFileField(alias="PowerReference", field_name="wtc_PowerRef_endvalue", table_name="tblSCTurbine"), +] + +# The Hill of Towie source-native column schema the synthetic pipeline and methods see. The +# raw 10-min tag names (``wtc_*``) are kept as-is (no v0 aliasing); the long-format turbine +# identifier is ``TurbineName`` (assigned by :func:`scada_wide_to_long`). Built directly from +# the source-native tag constants above, so the schema carries no v0 vocabulary. +HOT_TURBINE_COL = "TurbineName" +HOT_COLUMNS = ColumnSchema( + turbine=HOT_TURBINE_COL, + active_power=_TAG_ACTIVE_POWER_MEAN, + active_power_min=_TAG_ACTIVE_POWER_MIN, + wind_speed=_TAG_WIND_SPEED_MEAN, + wind_speed_sd=_TAG_WIND_SPEED_SD, + gen_rpm=_TAG_GEN_RPM_MEAN, + pitch=_TAG_PITCH_MEAN, + reactive_power=_TAG_REACTIVE_POWER_MEAN, + nacelle_position=_TAG_NACELLE_POSITION_MEAN, + ambient_temp=_TAG_AMBIENT_TEMP_MEAN, + availability=_TAG_AVAILABILITY, +) + +# Baseline rated power of the Hill of Towie test turbines (kW); matches the synthetic generator's +# baseline ``rated_power_kw`` default and caps the power-model counterfactual predictions. +HOT_RATED_POWER_KW = 2300.0 + +# Hub height of the Hill of Towie turbines (m); feeds the ERA5 hub-height wind-speed derivation. +HOT_HUB_HEIGHT_M = 59.0 + +# The reference active-power companion statistics (max/min/SD) a method may opt into as features. +HOT_ACTIVE_POWER_STAT_COLS: tuple[str, ...] = ( + _TAG_ACTIVE_POWER_MAX, + _TAG_ACTIVE_POWER_MIN, + _TAG_ACTIVE_POWER_SD, +) + + +def _unpack_hot_10min_year( + *, + data_dir: Path, + year: int, + serial_numbers: Sequence[int], + fields_to_load: Sequence[WPSBackupFileField], +) -> pd.DataFrame: + """Unpack one full year zip into a wide, serial-keyed 10-min dataframe (the slow step). + + This is the expensive part of :func:`load_hot_10min_data` (reading and pivoting every monthly + CSV in the year zip); it is cached per (year, turbine) by :func:`_load_unpacked_hot_10min`. + The result covers the whole year (no window clipping), is keyed on serial numbers (not turbine + names), and only includes the requested ``serial_numbers``. Columns keep their source-native + tag names (the ``wtc_*`` field names); any v0 aliasing is the v0 baseline's concern. + """ + from tqdm import tqdm # noqa: PLC0415 (lazy: keep network deps out of the import path) + + tables_to_load = {x.table_name for x in fields_to_load} + zip_path = data_dir / f"{year}.zip" + logger.info("Beginning 10min data unpacking: %s", zip_path) + with ZipFile(zip_path) as zip_file: + year_dfs = [] + for _table in tqdm(tables_to_load, desc=f"unpacking {zip_path.stem}"): + table_dfs = [] + for _month in range(1, 13): + if (fname := f"{_table}_{year}_{_month:02d}.csv") not in zip_file.namelist(): + continue + _df = pd.read_csv(zip_file.open(fname), index_col=0, parse_dates=True)[ + ["StationId", *[x.field_name for x in fields_to_load if x.table_name == _table]] + ] + if _df.index.name != "TimeStamp": + msg = f"unexpected index name, {_df.index.name =}" + raise ValueError(msg) + if not isinstance(_df.index, pd.DatetimeIndex): + _df.index = pd.to_datetime(_df.index, format="ISO8601") + if not isinstance(_df.index, pd.DatetimeIndex): + msg = f"unexpected index type, {_df.index.name =} {type(_df.index)=}" + raise TypeError(msg) + # convert to Start Format UTC + _df.index = _df.index.tz_localize("UTC") # type:ignore[attr-defined] + _df.index = _df.index - pd.Timedelta(minutes=10) + _df.index.name = "TimeStamp_StartFormat" + # drop any timestamps not in this month; files overlap by 10 minutes + _df = _df[(_df.index.year == year) & (_df.index.month == _month)] # type:ignore[attr-defined,assignment] + _df = _df[_df["StationId"].isin(serial_numbers)] + pivoted_df = _df.pivot_table( + index=_df.index.name, + columns="StationId", + values=[x for x in _df.columns if x != "StationId"], + ).swaplevel(axis=1) + table_dfs.append(pivoted_df) + table_df = pd.concat(table_dfs, verify_integrity=True, sort=True) + year_dfs.append(table_df) + return pd.concat(year_dfs, axis=1) + + +def _year_turbine_cache_path( + *, + year: int, + serial_number: int, + fields_to_load: Sequence[WPSBackupFileField], + cache_dir: Path, +) -> Path: + """Build the deterministic parquet path for one (year, turbine). + + The Zenodo record is a fixed one-zip-per-year layout, so a turbine-year is the stable unit of + work: the path depends only on the year, the turbine, and the field set (folded into a short + hash so a custom field selection gets its own files). It deliberately does **not** depend on + the requested window or the rest of the turbine subset, so any study reuses these files. + """ + fields_blob = json.dumps( + {"fields": sorted(f"{x.table_name}.{x.field_name}->{x.alias}" for x in fields_to_load)}, + sort_keys=True, + ) + fields_hash = hashlib.sha256(fields_blob.encode("utf-8")).hexdigest()[:16] + wtg_number = serial_number - _HOT_SERIAL_OFFSET + return cache_dir / f"hot10min_{year}_T{wtg_number:02d}_{fields_hash}.parquet" + + +def _load_unpacked_hot_10min( + *, + data_dir: Path, + years_to_load: Sequence[int], + serial_numbers: Sequence[int], + fields_to_load: Sequence[WPSBackupFileField], + cache_dir: Path, +) -> pd.DataFrame: + """Return the full-year, serial-keyed 10-min df for the requested years and turbines. + + Caches one parquet per (year, turbine): a turbine-year is unpacked from its zip at most once + and then reused for any window or turbine subset. When some requested turbines are not yet + cached for a year, that year's zip is unpacked once for just those turbines and one parquet is + written per turbine. Delete a file to force a re-unpack of that turbine-year. + """ + cache_dir.mkdir(parents=True, exist_ok=True) + year_frames = [] + for year in years_to_load: + paths = { + serial: _year_turbine_cache_path( + year=year, + serial_number=serial, + fields_to_load=fields_to_load, + cache_dir=cache_dir, + ) + for serial in serial_numbers + } + missing = [serial for serial in serial_numbers if not paths[serial].exists()] + if missing: + logger.info("HoT %d cache miss for turbines %s; unpacking", year, missing) + unpacked = _unpack_hot_10min_year( + data_dir=data_dir, + year=year, + serial_numbers=missing, + fields_to_load=fields_to_load, + ) + for serial in missing: + serial_df = unpacked.loc[:, unpacked.columns.get_level_values(0) == serial] + logger.info("Writing HoT cache: %s", paths[serial]) + serial_df.to_parquet(paths[serial]) + per_turbine = [pd.read_parquet(paths[serial]) for serial in serial_numbers] + year_frames.append(pd.concat(per_turbine, axis=1)) + return pd.concat(year_frames, verify_integrity=True, sort=True) + + +def load_hot_10min_data( + *, + data_dir: Path, + wtg_numbers: Sequence[int], + start_dt: pd.Timestamp, + end_dt_excl: pd.Timestamp, + use_turbine_names: bool = True, + custom_fields: Sequence[WPSBackupFileField] | None = None, + cache_dir: Path | None = None, +) -> pd.DataFrame: + """Return a wide 10-min SCADA dataframe for Hill of Towie (downloading year zips). + + Columns keep their source-native ``wtc_*`` tag names; the level-0 turbine key is the serial + number, or the ``T01``-style turbine name when ``use_turbine_names``. + + The slow zip-unpacking step is cached as parquet under ``cache_dir`` (defaults to + ``data_dir / "unpacked_cache"``), one file per (year, turbine). Because the Zenodo record is a + fixed one-zip-per-year layout, a turbine-year is unpacked at most once and then reused for any + window or turbine subset, so repeated studies over the same data skip re-reading every monthly + CSV. + """ + if str(start_dt.tz) != "UTC" or str(end_dt_excl.tz) != "UTC": + msg = "start_dt and end_dt_excl must be in UTC" + raise ValueError(msg) + if end_dt_excl <= start_dt: + msg = "end_dt_excl must be after start_dt" + raise ValueError(msg) + + serial_numbers = [x + _HOT_SERIAL_OFFSET for x in wtg_numbers] + first_year_to_load = start_dt.year + last_year_to_load = (end_dt_excl - pd.Timedelta(seconds=TIMEBASE_S)).year + years_to_load = list(range(first_year_to_load, last_year_to_load + 1)) + ensure_hot_data_files([f"{y}.zip" for y in years_to_load], data_dir=data_dir) + fields_to_load = hill_of_towie_fields if custom_fields is None else custom_fields + combined_df = _load_unpacked_hot_10min( + data_dir=data_dir, + years_to_load=years_to_load, + serial_numbers=serial_numbers, + fields_to_load=fields_to_load, + cache_dir=cache_dir if cache_dir is not None else data_dir / "unpacked_cache", + ) + if use_turbine_names: + cols = combined_df.columns + serial_to_name = {x: f"T{x - _HOT_SERIAL_OFFSET:02d}" for x in cols.get_level_values(0).unique()} + combined_df.columns = cols.set_levels( # type:ignore[attr-defined] + [serial_to_name[x] for x in cols.levels[0]], # type:ignore[attr-defined] + level=0, + ) + return ( + combined_df[(combined_df.index >= start_dt) & (combined_df.index < end_dt_excl)] + .resample(pd.Timedelta(seconds=TIMEBASE_S)) + .first() + ) + + +def calc_shutdown_duration(wind_up_df: pd.DataFrame) -> pd.DataFrame: + """Add a ``ShutdownDuration`` column and return the wind-up dataframe. + + Downtime is the time *not* ready to operate in the period; additionally, stuck data — + a turbine whose signals are unchanged from its own previous record (implausible, and a + sign of frozen/wrong telemetry) — above a low-wind threshold is treated as a full + period of downtime. + """ + wind_up_df = wind_up_df.copy() + wind_up_df[DataColumns.shutdown_duration] = TIMEBASE_S - wind_up_df["Time ready to operate in period"].fillna( + TIMEBASE_S + ) + signal_cols = [ + DataColumns.active_power_mean, + DataColumns.active_power_sd, + DataColumns.wind_speed_mean, + DataColumns.wind_speed_sd, + DataColumns.gen_rpm_mean, + DataColumns.pitch_angle_mean, + DataColumns.yaw_angle_mean, + ] + # Stuck (frozen) telemetry: every signal unchanged from the turbine's OWN previous + # record. The frame holds one row per (timestamp, turbine) interleaved by timestamp, + # so both the forward-fill and the diff must be grouped by turbine — an ungrouped diff + # would compare adjacent rows belonging to different turbines, not a turbine over time. + diffdf = ( + wind_up_df.groupby("TurbineName", observed=False)[signal_cols] + .ffill() + .fillna(0.0) + .groupby(wind_up_df["TurbineName"], observed=False) + .diff() + ) + stuck_data = (diffdf == 0).all(axis=1) + very_low_wind_threshold = 1.5 + very_low_wind = wind_up_df[DataColumns.wind_speed_mean] < very_low_wind_threshold + stuck_filter = stuck_data & (~very_low_wind) + wind_up_df.loc[stuck_filter, DataColumns.shutdown_duration] = TIMEBASE_S + return wind_up_df + + +def scada_wide_to_long(scada_df: pd.DataFrame, *, columns: ColumnSchema = HOT_COLUMNS) -> pd.DataFrame: + """Convert wide two-level ``scada_df`` to a narrow, source-native long frame. + + ``scada_df`` has two column levels (turbine, field); the result has one column level plus a + ``columns.turbine`` identifier column, and keeps the source-native ``wtc_*`` field names. This + is the method-facing layout: v0-specific aliasing and the derived ``PitchAngleMean`` / + ``ShutdownDuration`` columns are added later by :func:`long_to_wind_up_format`, which only the + v0 baseline needs. + """ + # future_stack=True only exists in pandas >= 2.1; without it >= 2.1 emits a + # FutureWarning (an error under the test config). Fall back for pandas 2.0.x. + try: + stacked = scada_df.stack(level=0, future_stack=True) # noqa: PD013 + except TypeError: + stacked = scada_df.stack(level=0, dropna=False) # noqa: PD013 + return stacked.reset_index(level=1).rename(columns={"StationId": columns.turbine}) + + +def long_to_wind_up_format(long_df: pd.DataFrame) -> pd.DataFrame: + """Convert a source-native long frame (see :func:`scada_wide_to_long`) to wind-up format. + + Renames the Hill of Towie ``wtc_*`` tag names to their v0 :class:`DataColumns` aliases, derives + ``PitchAngleMean`` from the three per-blade pitch columns when absent, and computes + ``ShutdownDuration``. This is the v0 baseline's on-ramp; the rest of the pipeline never needs it. + """ + alias_by_field = {f.field_name: f.alias for f in hill_of_towie_fields} + wind_up_df = long_df.rename(columns=alias_by_field) + if DataColumns.pitch_angle_mean not in wind_up_df.columns: + wind_up_df[DataColumns.pitch_angle_mean] = wind_up_df[["pitch_angle_a", "pitch_angle_b", "pitch_angle_c"]].mean( + axis=1 + ) + return calc_shutdown_duration(wind_up_df) + + +# -------------------------------------------------------------------------------------- +# Turbine metadata + top-level loader +# -------------------------------------------------------------------------------------- +def load_hot_metadata(*, data_dir: Path | None = None, wtg_names: Sequence[str] | None = None) -> pd.DataFrame: + """Load Hill of Towie turbine metadata (Name, Latitude, Longitude) in wind-up format.""" + data_dir = data_dir if data_dir is not None else get_data_dir() + ensure_hot_data_files(["Hill_of_Towie_turbine_metadata.csv"], data_dir=data_dir) + metadata_path = data_dir / "Hill_of_Towie_turbine_metadata.csv" + logger.info("Reading: %s", metadata_path) + return_df = ( + pd.read_csv(metadata_path) + .loc[:, ["Turbine Name", "Latitude", "Longitude"]] + .rename(columns={"Turbine Name": "Name"}) + .assign(TimeZone="UTC", TimeSpanMinutes=10, TimeFormat="Start") + ) + if wtg_names is not None: + # only return return_df rows where the turbine name is in wtg_names + return_df = return_df[return_df["Name"].isin(wtg_names)] + return return_df + + +def load_hot_scada( + *, + start_dt: pd.Timestamp, + end_dt_excl: pd.Timestamp, + wtg_numbers: Sequence[int] | None = None, + wtg_names: Sequence[str] | None = None, + data_dir: Path | None = None, +) -> tuple[pd.DataFrame, pd.DataFrame]: + """Download (if needed) and load source-native long Hill of Towie SCADA plus metadata. + + Downloads and caches the v2 datapack year zips from Zenodo, unpacks the requested window, and + reshapes to a long frame with source-native ``wtc_*`` tag names (see :data:`HOT_COLUMNS`). + Returns ``(scada_df, metadata_df)`` ready for the synthetic generator. v0-specific aliasing is + applied later, only by the v0 baseline (see :func:`long_to_wind_up_format`). + + :param start_dt: inclusive UTC window start + :param end_dt_excl: exclusive UTC window end + :param wtg_numbers: turbine numbers to load; defaults to all (1..21) + :param data_dir: data/cache directory; defaults to :func:`get_data_dir` + """ + data_dir = data_dir if data_dir is not None else get_data_dir() + metadata_df = load_hot_metadata(data_dir=data_dir, wtg_names=wtg_names) + wtg_numbers = list(range(HOT_FIRST_WTG, HOT_LAST_WTG + 1)) if wtg_numbers is None else list(wtg_numbers) + wide_scada_df = load_hot_10min_data( + data_dir=data_dir, + wtg_numbers=wtg_numbers, + start_dt=start_dt, + end_dt_excl=end_dt_excl, + ) + scada_df = scada_wide_to_long(wide_scada_df) + return scada_df, metadata_df diff --git a/benchmarking/synthetic/upgrades.py b/benchmarking/synthetic/upgrades.py new file mode 100644 index 00000000..8a06c271 --- /dev/null +++ b/benchmarking/synthetic/upgrades.py @@ -0,0 +1,226 @@ +"""Synthetic turbine-upgrade callables and their resolution. + +An upgrade is a callable ``(rows) -> UpgradeEffect`` describing how it changes a test +turbine's treated rows. Upgrades compose: ``apply_upgrades`` resolves a list against the +*original* baseline in a defined order (Cp ratios multiply and are applied through the +Cp core, then a rated-power change, then a nacelle wind-speed change), so the result does +not depend on list order in surprising ways. + +Phase 1 implements the four profiles Issue 1 names: constant, wind-speed-dependent and +condition (turbulence-intensity) dependent Cp changes, and rated-power change. Pitch, +rpm, yaw and wake-steering upgrades are future work. +""" + +from __future__ import annotations + +from dataclasses import dataclass +from typing import TYPE_CHECKING + +import numpy as np +import numpy.typing as npt + +from benchmarking.synthetic.cp_core import power_from_cp_change, region2_fraction, rpm_from_power_change +from benchmarking.synthetic.sources.hill_of_towie import HOT_COLUMNS + +if TYPE_CHECKING: + import pandas as pd + + from benchmarking.synthetic.cp_core import CpCore + from benchmarking.synthetic.schema import ColumnSchema + + +@dataclass +class UpgradeEffect: + """One upgrade's contribution, resolved against the original baseline rows. + + :param cp_ratio: per-row (or scalar) multiplicative Cp ratio, e.g. 1.02 for +2% Cp + :param ws_factor: multiplicative change to the nacelle wind speed, e.g. 1.01 for +1% + :param new_rated_power_kw: rated-power override, or None to leave rated unchanged + """ + + cp_ratio: npt.ArrayLike = 1.0 + ws_factor: float = 1.0 + new_rated_power_kw: float | None = None + + +@dataclass(frozen=True) +class ConstantCpChange: + """A flat Cp change in region 2 (e.g. blade cleaning, fouling or add-ons).""" + + delta: float + ws_delta: float = 0.0 + + @property + def description(self) -> dict: + """Return serialisable provenance describing this upgrade.""" + return {"kind": "constant_cp", "delta": self.delta, "ws_delta": self.ws_delta} + + def __call__(self, rows: pd.DataFrame, columns: ColumnSchema) -> UpgradeEffect: # noqa: ARG002 + """Return this upgrade's effect on the given rows.""" + return UpgradeEffect(cp_ratio=1.0 + self.delta, ws_factor=1.0 + self.ws_delta) + + +@dataclass(frozen=True) +class WindSpeedCpChange: + """A wind-speed-dependent Cp change (e.g. the AeroUp region-2 shape). + + The Cp delta is interpolated over ``ws_points`` against the turbine's original + wind speed; outside the point range the nearest endpoint delta is held. + """ + + ws_points: tuple[float, ...] + deltas: tuple[float, ...] + ws_delta: float = 0.0 + + @property + def description(self) -> dict: + """Return serialisable provenance describing this upgrade.""" + return { + "kind": "wind_speed_cp", + "ws_points": list(self.ws_points), + "deltas": list(self.deltas), + "ws_delta": self.ws_delta, + } + + def __call__(self, rows: pd.DataFrame, columns: ColumnSchema) -> UpgradeEffect: + """Return this upgrade's effect on the given rows.""" + original_ws = rows[columns.wind_speed].to_numpy(dtype=float) + delta = np.interp(original_ws, self.ws_points, self.deltas) + return UpgradeEffect(cp_ratio=1.0 + delta, ws_factor=1.0 + self.ws_delta) + + +def _condition_series(rows: pd.DataFrame, by: str, columns: ColumnSchema) -> npt.NDArray[np.float64]: + """Compute a treatment-invariant condition signal from the original rows. + + ``by="ti"`` is turbulence intensity (wind-speed SD / wind-speed mean); any other value is + treated as the name of a column already present in ``rows``. + """ + if by == "ti": + ws = rows[columns.wind_speed].to_numpy(dtype=float) + sd = rows[columns.wind_speed_sd].to_numpy(dtype=float) + # NaN (not inf/0-division warning) for calm rows; warnings are errors in tests. + return np.divide(sd, ws, out=np.full_like(sd, np.nan), where=ws != 0) + return rows[by].to_numpy(dtype=float) + + +@dataclass(frozen=True) +class ConditionCpChange: + """A Cp change that varies with a condition signal of the original data. + + Phase 1 supports ``by="ti"`` (turbulence intensity = WindSpeedSD / WindSpeedMean); + the Cp delta is interpolated over ``points`` of that condition. ``by`` may also name + any column already present in the rows. + """ + + by: str + points: tuple[float, ...] + deltas: tuple[float, ...] + ws_delta: float = 0.0 + + @property + def description(self) -> dict: + """Return serialisable provenance describing this upgrade.""" + return { + "kind": "condition_cp", + "by": self.by, + "points": list(self.points), + "deltas": list(self.deltas), + "ws_delta": self.ws_delta, + } + + def __call__(self, rows: pd.DataFrame, columns: ColumnSchema) -> UpgradeEffect: + """Return this upgrade's effect on the given rows.""" + condition = _condition_series(rows, self.by, columns) + delta = np.interp(condition, self.points, self.deltas) + return UpgradeEffect(cp_ratio=1.0 + delta, ws_factor=1.0 + self.ws_delta) + + +@dataclass(frozen=True) +class RatedPowerChange: + """A change to the turbine's rated power. + + A downrate (new < old) caps power at the new rated and leaves region-2 power + unchanged. An uprate (new > old) lifts the region-3 fraction of power toward the new + rated (deep region-2 power is essentially unaffected), then caps at the new rated. + The uprate model is a documented synthetic choice rather than measured physics. + """ + + new_rated_power_kw: float + + @property + def description(self) -> dict: + """Return serialisable provenance describing this upgrade.""" + return {"kind": "rated_power", "new_rated_power_kw": self.new_rated_power_kw} + + def __call__(self, rows: pd.DataFrame, columns: ColumnSchema) -> UpgradeEffect: # noqa: ARG002 + """Return this upgrade's effect on the given rows.""" + return UpgradeEffect(new_rated_power_kw=self.new_rated_power_kw) + + +def apply_rated_change( + power_kw: npt.NDArray[np.float64], + *, + original_power_kw: npt.NDArray[np.float64], + old_rated_kw: float, + new_rated_kw: float, +) -> npt.NDArray[np.float64]: + """Apply a rated-power change to (already Cp-adjusted) power. + + In future this function should change other test turbine fields especially pitch angle but this is deferred for now. + + See :class:`RatedPowerChange` for the downrate/uprate behaviour. + """ + if new_rated_kw <= old_rated_kw: + return np.minimum(power_kw, new_rated_kw) + region3_fraction = 1.0 - region2_fraction(original_power_kw) + ratio = new_rated_kw / old_rated_kw + lifted = power_kw * (1.0 + region3_fraction * (ratio - 1.0)) + return np.minimum(lifted, new_rated_kw) + + +def apply_upgrades( + rows: pd.DataFrame, upgrades: list, *, cp: CpCore, columns: ColumnSchema = HOT_COLUMNS +) -> pd.DataFrame: + """Resolve and apply a list of upgrades to one test turbine's treated rows. + + Cp ratios from all upgrades multiply together and are applied to the original power + through the Cp core (region-2 weighted, rated-clipped); rpm tracks the resulting + power change and the nacelle wind speed is scaled by the combined ws factor. The + input ``rows`` is not mutated. + + :param rows: treated rows for one test turbine, carrying the original SCADA tags + :param upgrades: upgrade callables to resolve + :param cp: the turbine's Cp core (rated power and Cp parameters) + :param columns: the source-native column schema ``rows`` is keyed by + :return: a new frame with modified power, rpm and wind-speed columns + """ + out = rows.copy() + n = len(rows) + baseline_power = rows[columns.active_power].to_numpy(dtype=float) + baseline_rpm = rows[columns.gen_rpm].to_numpy(dtype=float) + baseline_ws = rows[columns.wind_speed].to_numpy(dtype=float) + + cp_ratio = np.ones(n) + ws_factor = 1.0 + new_rated_power_kw: float | None = None + for upgrade in upgrades: + effect = upgrade(rows, columns) + cp_ratio = cp_ratio * np.asarray(effect.cp_ratio, dtype=float) + ws_factor *= effect.ws_factor + if effect.new_rated_power_kw is not None: + new_rated_power_kw = effect.new_rated_power_kw + + new_power = power_from_cp_change(baseline_power, cp_ratio=cp_ratio, rated_power_kw=cp.rated_power_kw) + if new_rated_power_kw is not None: + new_power = apply_rated_change( + new_power, + original_power_kw=baseline_power, + old_rated_kw=cp.rated_power_kw, + new_rated_kw=new_rated_power_kw, + ) + new_rpm = rpm_from_power_change(baseline_rpm=baseline_rpm, baseline_power_kw=baseline_power, new_power_kw=new_power) + + out[columns.active_power] = new_power + out[columns.gen_rpm] = new_rpm + out[columns.wind_speed] = baseline_ws * ws_factor + return out diff --git a/docs/v1/README.md b/docs/v1/README.md new file mode 100644 index 00000000..52a8fef9 --- /dev/null +++ b/docs/v1/README.md @@ -0,0 +1,36 @@ +# wind-up v1 + +This folder holds the planning material for the **v1** major upgrade of wind-up. +v1 is developed on the `v1` branch; feature PRs target `v1` rather than `main`. + +## Contents + +- **[goals.md](goals.md)** — the north-star vision and goals for v1 (the "why"). +- **[roadmap.md](roadmap.md)** — workstreams (epics), phasing, and how the work is + managed (the "what" and "in what order"). +- **[issues.md](issues.md)** — drafts of the first concrete issues. These are + refined here before being created as GitHub issues. +- **[references.md](references.md)** — related open-source tools (FLASC, OpenOA, + DSWE) and key methodology references (Kanev TNO report) to investigate later. + +## One-paragraph summary + +v0 is a single-method tool: it measures turbine-upgrade uplift with a binned +power-curve, test-vs-reference method. v1 turns wind-up into a **platform** in +which **alternative uplift methods are pluggable and objectively benchmarked** on +synthetic datasets with known ground truth. The driving goals are *more accurate +results from shorter campaigns* and *richer conditional information* about how an +upgrade performs (in wakes vs free-stream, day vs night, by direction/stability). +The first wave of work is **methodology-first**: build the public evaluation +harness, wire the v0 method in as the baseline to beat, and develop the first new +candidate method — judged on **P50 accuracy and precision only**. An uncertainty +(P95) model is deferred until a winning P50 method is found. + +## How this work is managed + +- **Branch:** `v1` (published). PRs target `v1`. +- **Methodology prototyping:** hybrid model — the synthetic-data + evaluation + harness is **public** (suitable as a WeDoWind exercise / open benchmark, in the + spirit of the Hill of Towie Kaggle challenge). Individual candidate methods may + be prototyped anywhere, and are **ported into wind-up v1 only once they beat the + v0 baseline** on the harness. diff --git a/docs/v1/findings.md b/docs/v1/findings.md new file mode 100644 index 00000000..a2c448df --- /dev/null +++ b/docs/v1/findings.md @@ -0,0 +1,2056 @@ +# wind-up v1 — findings log + +Empirical findings from the v1 benchmarking work. Newest first. Each entry records what was +observed, the evidence, the root cause, and what (if anything) it implies for the method design +or the issues list. Keep entries reproducible: name the study driver and the diagnostics they came +from, not just conclusions. + +--- + +## F33 — A per-bin cell with 1-2 records reports a **confidently wrong** sigma (coverage 0.158, and one cell reported sigma **exactly 0** while being 14 pp out). Fixed with a t-inflated per-record-scatter fallback, *selected* below a 3-record floor (not blended) so the calibrated bootstrap regime is untouched + +*2026-07-16. Prompted by the question "what does a bin with only one or two data points do?" — a case +F29/F31/F32 had all bucketed away. Reproduce: the 256-replicate `cases.csv`, per-bin cells binned on +`min(n_upgraded_records, n_baseline_records)`.* + +**The failure.** Measured coverage by the cell's *thinner* side (either side starves the ratio): + +| min(n_on, n_off) | 1 | 2 | 3-4 | 5-7 | 8-11 | 12-20 | 21-50 | >50 | +|---|---|---|---|---|---|---|---|---| +| coverage | **0.158** | **0.237** | 0.579 | 0.656 | 0.718 | 0.724 | 0.676 | 0.680 | +| SE from target | **-3.0** | **-2.5** | -0.7 | -0.2 | +0.3 | +0.6 | -0.2 | -0.3 | +| median sigma [pp] | 0.90 | 2.34 | 3.82 | 4.88 | 4.90 | 3.95 | 3.01 | 1.07 | +| median \|error\| [pp] | 7.18 | 4.16 | 2.77 | 2.89 | 2.31 | 2.09 | 1.78 | 0.63 | + +At one record per side sigma is **8x too small**; **8 cells reported sigma exactly 0.0** with a median +error of 10 pp and a worst of 22.6 pp. A reader would see "uplift -14.3% ± 0.0%". + +**Why, and why it is not merely imprecision.** The bootstrap resamples whole **blocks**. If a cell's +records sit inside one block, a resample either includes that block `k` times or not at all — and the +ratio is `k*test / k*ref`, which is **independent of `k`**. Every finite resample returns the +identical uplift, so the spread is zero. The bootstrap does not lose precision here; it reports +certainty. That is the worst possible failure mode for a number whose entire job is to say how much +to trust an estimate. + +**The fix: a fallback, *selected* below a floor — not NaN, not a blend.** The path here is worth +recording because two earlier versions were wrong in instructive ways. + +*First attempt — report NaN below the floor.* Rejected on the user's push: "NaN is not good enough". +Correct. "This bin has too little data to quantify" is more honest than "±0.0%", but a *number* the +consumer can act on is better than either, and there is a real per-record scatter to estimate from. + +*The estimator.* `sigma_fallback = s_rel * sqrt(1/n_on + 1/n_off) * t.ppf(norm.cdf(1), df)`, where +`s_rel` is the campaign's own relative scatter about its test/reference ratio (a ratio of sums, so it +does not explode near cut-in), and the `scipy.stats.t` multiplier is `wind_up`'s own convention from +`pp_analysis` — the 1-sigma-equivalent quantile, `-> 1.0` as data grows, keyed off the *thinner* side +via `clip(lower=2)`. The `sqrt(1/n_on + 1/n_off)` shape is the load-bearing part (the `t` factor is +only 1.84x at df=1, against the 8x shortfall); it is validated directly — implied `s_rel` is constant +(~12-17%) across a 13x range of that shape statistic. + +*Second attempt — `max(bootstrap, fallback)` everywhere.* Rejected, on evidence and on the user's +guidance ("avoid `max` where the bootstrap performs well; it might mess up another farm"). Measured +under `max`, the well-covered regime **degrades**: solid-bin (min>10) coverage 0.681 -> 0.770, +headline 0.682 -> 0.733. The fallback is comparable to the bootstrap even at n~100 (`s_rel` ~15%), so +`max` inflates cells that were already right — and a `max` tuned here could over-inflate elsewhere. + +*Third attempt — hard selection at a floor.* Bootstrap where `min(n_on, n_off) >= 3`, else fallback. +This is `below_floor` in the pre-registered comparison and leaves the covered regime **untouched** +(solid 0.681, headline 0.682, to three decimals). But it swaps a ~7.5x **cliff** in the reported +sigma at the boundary (median 36 pp at min 2 -> 4.8 pp at min 3), which the user objected to. + +*The rule that ships — a linear ramp across the floor.* The report is +`w*bootstrap + (1-w)*fallback` with `w = clip((min_side - 3)/4, 0, 1)`: pure fallback at +`min_side <= 3`, pure bootstrap at `>= 7`, and `1/4, 1/2, 3/4` at 4, 5, 6. Constants +`_BLEND_LO_RECORDS = 3`, `_BLEND_HI_RECORDS = 7`. The endpoints are not arbitrary: the ramp *starts* +where the bootstrap first becomes computable (min 3) and *ends* where it becomes trustworthy — its +own coverage is ~0.58-0.62 through min 3-6 and only reaches ~0.68+ by min 7. So the fallback carries +the cell through exactly the band where the bootstrap under-covers. + +| metric | hard selection | **linear ramp (shipped)** | +|---|---|---| +| solid (min>10) coverage | 0.681 | **0.681** (pure bootstrap) | +| headline coverage | 0.682 | **0.682** (pure bootstrap) | +| cliff at min 2->3 (sigma ratio) | 7.5x | **1.8x** | +| min 3->7 sigma ramp (pp) | 36 -> 4.8 (jump) | **20 -> 14 -> 10 -> 7.6 -> 4.3** | + +The sparse cells still **over**-cover (~1.0 at min<=2, where only the wide fallback exists). That is +the safe direction — a too-wide uncertainty on a one-point bin is honest, where the old sigma=0 was a +lie — and the price of having no better estimator there. + +The `1/4, 1/2, 3/4` shape is the user's own default ("no strong data guidance, so ramp at ∓1 around +the threshold"); the data set the *placement* (LO/HI = the bootstrap's computable and trustworthy +points) rather than the shape. + +**Nothing precludes a zero sigma** (the user's separate philosophical point, and correct): on perfect +data `s_rel -> 0` and the resamples agree, so both components -> 0 and the report is 0. F31 found no +irreducible floor to justify banning it. An interim version NaN-ed a zero spread to trap the +1-record artefact; that punished the legitimate zero to catch the artefact, and the fallback removes +the need — the artefact is a sparse cell, so the floor sends it to the fallback before the zero ever +surfaces. + +**Both components are emitted per cell** (`sigma_bootstrap`, `sigma_fallback`), so the selection rule +itself can be re-judged from a saved sweep without re-running one — as it was here. + +**Why the earlier rounds missed it, which is the more useful lesson.** F29/F31/F32 all bucketed +per-bin coverage at `(0, 30]` records, which averages to a reassuring **0.623** and hides a 0.158 +subset. Worse, `calibration_summary` requires `sigma > 0`, so it counted the exact-zero cells as +`n_unusable` and **excluded the very worst cases from the coverage metric**. `n_unusable` was reported +in every table and never analysed. **A metric that drops its own pathological cases will always look +calibrated** — check the sparsest bucket explicitly. + +**Scope:** the ramp changes the reported sigma only on per-bin cells with `min_side < 7` (a few +percent), zero headline cells, and no uplift (verified: the compare reports both methods UNCHANGED). + +--- + +## F32 — At 256 replicates the bootstrap-only sigma is **calibrated**: pooled coverage 0.682 vs a 0.683 target at 6h blocks, every campaign length from 1 week to 1 year within 0.5 SE. No further uncertainty component is justified, and the 6h default is confirmed rather than merely safe + +*2026-07-16 (toggle-specialist uncertainty, round 3 — the power run F31 asked for). Reproduce: +`uv run python -m benchmarking.baselines.study_toggle_specialist_uncertainty --replicates 256 +--block-hours 1 3 6` — 96,768 cells. **The decision rule below was fixed before looking at the +data**, so this is a test rather than a search for a justification.* + +**The question F31 left open.** Coverage sat consistently at ~0.65 against 0.683 — never significant +(-1.26 SE), never above target either. That is the shape of either a real ~5% optimism or noise, and +64 replicates could not tell them apart. Guessing would have meant fitting noise; the fix is power, +not modelling. + +**The decision rule, pre-registered:** an inflation `k` is justified only if it is *consistent* — it +must bring **every** campaign length closer to target (`n_worse == 0`) and lower the worst deviation. +Trading one length for another is not an improvement. + +**Result: the bootstrap-only sigma is calibrated.** Pooled over 1-52 weeks (~873 independent draws, +SE 0.016): + +| block | pooled coverage | SE from target | +|---|---|---| +| 1h | 0.758 | **+4.76** | +| 3h | 0.701 | +1.15 | +| **6h** | **0.682** | **-0.09** | + +Per campaign length at 6h: **0.689 / 0.684 / 0.698 / 0.668 / 0.663 / 0.689** for 1/2/4/8/26/52 weeks +— every one within **0.5 SE** of target, worst deviation **0.020**. `std(z)` agrees independently at +**1.013 / 0.975 / 0.950 / 1.028 / 1.041 / 0.927** (target 1.0), and `median|z|` at 0.61-0.74 against +the 0.674 a calibrated normal gives. Coverage and the magnitude-sensitive read concur. + +**No inflation is justified — `k = 1.0` is optimal.** It is the only value with `n_worse == 0`; every +`k >= 1.05` moves 4-6 of the 6 lengths away from target and raises the worst deviation from 0.020 to +0.045+. **F31's ~0.65 was noise**, confirmed at 4x the power. + +**The 6h default is confirmed, not merely safe.** F28 chose it on robustness grounds when 2h/3h/6h +were statistically indistinguishable. At 256 replicates they are distinguishable and 6h is right: +**1h over-covers significantly (+4.76 SE)** and 3h is drifting high (+1.15 SE). The robustness +argument (9 toggle cycles per block; enough blocks even at 1 week) picked the value the data now +independently endorses. + +**Where this leaves the uncertainty work.** The *scale* corrections anticipated at design time are +rejected on evidence — an inflation (above), the campaign-level systematic floor (F31), and the shape +correction (F31, itself an artefact). What remains is a plain circular block bootstrap at 6h blocks, +calibrated from one week to one year across placebo and +/-2% profiles. + +> **Corrected by F33.** This section originally concluded "the right move is to add nothing", +> including the low-count term. That was wrong, and wrong for an instructive reason: every coverage +> read here pools per-bin cells at `(0, 30]` records, which averages a broken 0.158 subset (1-2 +> records) into a reassuring 0.623 — and `calibration_summary` excludes `sigma <= 0` cells as +> `n_unusable`, so the metric structurally dropped the worst cases. A **fallback below a 3-record +> floor** *is* justified. The claim that survives is narrower and still holds: no correction is +> justified in the regime the bootstrap already covers — which is exactly why F33 ships a *selection* +> at the floor rather than a `max` that would have contaminated it. + +The residual caveats are honest and small: 52-week coverage rests on only ~16 independent draws +(SE 0.116), and ~1.6% of 1-week campaigns report a very large sigma where the reference denominator +nears zero — correct behaviour, but it makes `mean_sigma` a misleading summary (use medians or +coverage). + +--- + +## F31 — Extending the campaign grid to a year kills F29's leading hypothesis: there is **no campaign-level systematic floor**, and the platykurtosis was an artefact of a narrow start range. The bootstrap-only sigma keeps working on ample data, and **no further component is justified** + +*2026-07-15 (toggle-specialist uncertainty, round 2). Reproduce: +`uv run python -m benchmarking.baselines.study_toggle_specialist_uncertainty` — now 64 replicates x 3 +profiles x **1/2/4/8/26/52 weeks** x 7 block lengths (56,448 cells). Two config changes from F28/F29, +both deliberate: the grid reaches a year, and `min_pre_months=0` with the start range widened to the +whole dataset (2016-01-01..2020-01-01).* + +**Why the config changed.** `toggle_specialist` drops pre-campaign rows (`restrict_to_campaign`), so +`min_pre_months` buys it nothing and only costs start-range span — and span is exactly what long +campaigns need, because **replicates stop being independent once their windows overlap**. A 52-week +campaign is 364d, so the old 730d range held only ~2 non-overlapping positions. Widening to 1461d +doubles that. `independent_draws()` now reports the honest count per length and the SE is quoted on +it: **1-8wk ~64 draws (SE 0.058), 26wk ~32 (0.082), 52wk ~16 (0.116)**. A 52-week coverage anywhere +in **0.45..0.92** is indistinguishable from calibrated — long campaigns are precise but their +*uncertainty* is weakly evidenced, because 5 years of SCADA holds few independent year-long windows. + +**Long campaigns are the discriminating test, and they kill the floor.** F29's lead candidate was an +irreducible campaign-level systematic that a within-campaign bootstrap cannot see. `sigma_boot` +shrinks with data, so any such floor **must dominate** once sigma is small enough. It does not: + +| campaign | 1w | 2w | 4w | 8w | 26w | 52w | +|---|---|---|---|---|---|---| +| median sigma [pp] | 1.156 | 0.784 | 0.549 | 0.387 | 0.197 | **0.135** | +| median \|error\| [pp] | 0.733 | 0.476 | 0.310 | 0.216 | 0.157 | **0.110** | +| coverage | 0.599 | 0.646 | 0.698 | 0.724 | 0.594 | **0.635** | +| implied floor [pp] | 0.00 | 0.60 | 0.00 | 0.18 | 0.18 | **0.07** | + +At 52 weeks sigma is 0.135 pp. A 0.2 pp floor would swamp it and drive coverage to ~0.4; observed +**0.635**. The implied floor does not persist — it is *smaller* at 52wk (0.07) than at 8wk (0.18), +which is the signature of noise, not of a floor. **Hypothesis rejected.** Short campaigns could never +have decided this, because `sigma_boot` swamps any floor there. + +**The platykurtosis was an artefact of the narrow start range.** F29 measured kurtosis -0.35..-0.76 +(Shapiro p<0.05 everywhere) and read it as a real shape problem capping achievable coverage. With the +widened range it is **~0** (-0.25/-0.36/+0.04/-0.02/+0.08/-0.09). Drawing 64 windows from only 2 +years clustered them; the flat-topped error distribution was the clustering, not the estimator. +**Hypothesis rejected** — and a caution that F29's shape reasoning was over-read. + +**No count term, confirmed again.** Per-bin coverage by record-count decade at 6h blocks: +0.623 / 0.728 / 0.674 / 0.713 / 0.654 / 0.674 for `<=30 / 30-100 / 100-300 / 300-1k / 1k-3k / 3k-10k`. +No trend. + +**`mean_sigma` is a trap here; use medians or coverage.** 1-week `mean_sigma` reads 3.39 pp against an +RMS error of 1.64 pp — apparently 2x too wide — while coverage says 0.599, slightly too *narrow*. The +mean is dominated by **3 cases of 192 (1.6%)** with sigma up to **203 pp**. Those are not a defect: +they are 1-week campaigns whose reference denominator approaches zero in a low-wind week, and the +ratio estimator is genuinely unstable there, so sigma correctly says "no idea". They are also **not** +from the newly-added years (all three start in 2018), and 2016 campaigns look like every other year +(median 2668 used records, median sigma 0.414 pp, coverage 0.611) — the widening introduced no junk. + +**Block length matters less than F28 implied, once the grid reaches a year.** Pooled over 1-52wk +(~304 independent draws, SE 0.027): 1h **0.720**, 2h 0.657, 3h 0.661, 6h 0.649, 12h 0.656, 24h 0.649, +48h 0.641. Everything from 2h to 48h is flat within noise; only 1h stands out, by over-covering. F28's +sharp block-length gradient was real but **specific to a grid of only short campaigns**: at 26/52wk +`T/L` is large enough that block length is irrelevant, which dilutes it. The 6h default stands (F28's +short-campaign case is unaffected and 48h is still worst at 1 week, 0.573 vs 0.651), but the margin is +narrower than F28 suggested. + +**Verdict: no further uncertainty component is justified by this data.** Every candidate either fits +noise or fixes one campaign length while breaking another — a floor of 0.10 pp lifts pooled coverage +to 0.674 but pushes 52wk to 0.729 while leaving 1wk at 0.599; a 1.10x inflation reaches 0.686 pooled +but spreads 0.609..0.776 across lengths. Nothing anywhere is more than 1.6 SE from target. + +**The one open question is power, not modelling — and F32 settled it.** Coverage sits consistently at +~0.65 against 0.683 here: never significant, but never above target either, which is the shape of +either a real ~5% optimism or noise. This ensemble cannot tell them apart, so adding an inflation on +this evidence would be fitting noise. **F32 ran it at 256 replicates: it was noise** (pooled 0.682 at +6h blocks, -0.09 SE), and no inflation is justified. + +--- + +## F30 — `power_model`'s benchmark is **machine-specific** (~0.7 pp cross-machine, 14x its same-machine noise); `toggle_specialist`'s is portable to 5e-07 pp, so the toggle benchmark splits into a shared file plus one per platform + +*2026-07-15. Reproduce: `study_toggle_methods_compare` on a machine that did not record the +benchmark. The LightGBM behaviour is visible by fitting any `make_outcome_model` at `verbose=1`.* + +**Observation.** Running the toggle compare on the Linux laptop against a benchmark recorded on the +Windows laptop, `power_model` reports **MOVED — 25 of 84 cells, max ~0.70 pp** — permanently, and +regardless of the code under test. `toggle_specialist` reports UNCHANGED at **5e-07 pp** on the same +run, against the same foreign benchmark. + +**This is not run-to-run noise, and reading it as a stale baseline is wrong (I did).** The evidence +that looks damning: + +- two runs of identical code on one machine agree with **each other** to **0.045 pp** — matching the + ~0.05 pp the script documents; +- yet both deviate from the foreign benchmark by ~0.7 pp with their delta patterns correlated + **0.994**; +- and a clean `git archive HEAD` extract, containing no local changes at all, reproduces the same + 0.697 pp. + +That reads as "same deltas every run ⇒ deterministic ⇒ the code changed", and the conclusion drawn +was "the committed baseline does not reproduce on its own commit; regenerate it". **Wrong.** It is +deterministic **per machine**, not per code: both runs share this machine's LightGBM reduction order, +so both depart from the recording machine's identically. The disconfirming evidence was available and +ignored — the input data was three weeks stale and unchanged, and the recording predated the run by +40 minutes on nominally identical code. One question ("whose machine recorded it?") settled it. + +**Mechanism**, confirmed live rather than inferred: + +1. **LightGBM times the machine and picks its histogram strategy from the result** — + `[LightGBM] [Info] Auto-choosing col-wise multi-threading, the overhead of testing was 0.000365 + seconds.` `force_row_wise`/`force_col_wise` are unset in `_COMMON`, and row-wise and col-wise + accumulate gradient/hessian histograms in **different orders**. +2. **Floating-point addition is not associative**, so a different order changes the last bits. +3. **`num_threads` is unset**, defaulting to all cores (12 here), and the thread count partitions the + histogram reduction — another order change. +4. **Windows vs Linux** compounds it: different wheel, compiler (MSVC vs GCC), OpenMP runtime (vcomp + vs libgomp), SIMD codegen, libm. + +Then it **amplifies**: a split is an `argmax` over candidate gains, so a ~1e-16 difference flips a +near-tied split, changes a tree, and 600 boosted trees compound it into ~0.7 pp. +`toggle_specialist` is immune because it is a sum and a divide — no trees, no threads, no argmax. + +**Resolution: split the benchmark by portability, not by machine.** +`study_toggle_methods_compare_baseline.json` (v2) becomes three v3 files — `..._portable.json` +(`toggle_specialist`), `..._linux.json` and `..._win32.json` (`power_model`) — keyed on +`sys.platform`. A run diffs portable + this platform, merged. Each laptop writes only its own platform +file plus the shared one, so they never conflict in git. Portability is a per-method fact +(`_REPRODUCIBILITY`) sitting beside the band it already had, defaulting to `portable=False` because +wrongly claiming portability is a permanent, confusing failure on the other machine. + +**The portability invariant, and why the obvious version of it is wrong.** From one machine "the +numbers moved" is ambiguous: it means either the method changed or portability broke. Refusing on any +difference would block every legitimate re-record; a `--force` escape would just train the user past +the check. **The commit is the discriminator** — differ at the *same* commit ⇒ portability broke +(refuse); differ at a *different* commit ⇒ an accepted change (rewrite). The dirty-tree guard is what +makes the commit trustworthy enough to lean on. Two further subtleties, both found before they bit: + +- the comparison must use the **band**, not bit-equality: recorded cells carry wall time (differs + every run, so an exact test rewrites the shared file every recording and creates the very conflict + the split avoids), and `round(8)`'s 1e-8 resolution is only 2x the measured 5e-9 cross-machine + difference; +- the portable file's `git_commit` therefore records **when those numbers were last established**, + not who last ran a recording. An unchanged re-record leaves the file untouched, commit and all. + +**Provenance.** Each file now records `platform` / `cpu_count` / `python_version` / +`lightgbm_version`, and a mismatch warns (never fails). This is the direct fix for the hole above: +the file could not say where it came from, so the only way to find out was to ask a human. + +**`study_power_model_compare`'s benchmark: the numbers were never the problem.** A full read-only +re-run on the Linux laptop (the machine that records it) reproduces it exactly — **2030 of 2030 cells +neutral across both modes**. Its `e2e21b0-dirty` stamp is **very likely a false alarm**: its +`_git_commit()` counted **untracked** files as dirty, unlike `study_toggle_methods_compare`, which +deliberately passes `--untracked-files=no` because an untracked file cannot make a run irreproducible +from its commit — `git checkout ` would not have it. With a local `CLAUDE.md` sitting +untracked, that definition also made `--update-baseline` **impossible on this machine**, permanently. +Now aligned to the toggle script's definition, with tests pinning both directions. + +Three further guards ported to it, all previously absent: + +- it *labelled* a dirty tree but never **refused** one; +- `--accept-candidate` — the documented no-re-run accept, and so the likeliest route for a bad stamp + — promoted candidates without checking they were recorded clean; +- `record_baseline` read HEAD at *write* time, so an hours-long sweep straddling a commit would stamp + code that never ran. The commit is now captured before the sweep, as the toggle script already did. + +That script stays **single-machine by design**: every cell in it is `power_model`, so there is nothing +portable to split off, and it is always run on the Linux laptop. + +**Recording the Linux half of the toggle benchmark exercised the invariant for real, and it passed:** +`Portable baseline unchanged — portability confirmed, not rewritten`. `toggle_specialist`'s cells, +recorded on the Windows laptop, matched the Linux run within band, so the shared file was left +untouched — no churn, no conflict, and a live cross-machine confirmation rather than an assumption. + +--- + +## F29 — The anticipated failures of a bootstrap-only uncertainty did not appear: at a well-chosen block length `toggle_specialist`'s sigma is statistically indistinguishable from calibrated everywhere, and the sparse-bin failure was mostly F28's block-length artefact + +*2026-07-15 (toggle-specialist uncertainty, round 1). Reproduce: +`uv run python -m benchmarking.baselines.study_toggle_specialist_uncertainty` — 64 replicates x 3 +profiles x 1/2/4/8 weeks x 5 block lengths; the per-cell table is its `cases.csv` and the reads are +its `calibration_*.csv`. Coverage target 0.683; **binomial SE on 64 independent draws = 0.058**.* + +The prior going in was that a block bootstrap would work for long campaigns and well-populated bins +and fail for short campaigns and sparse bins, so a **data-count term** would be needed. Measured, at +6h blocks, headline coverage by campaign length is **1w 0.672, 2w 0.724, 4w 0.693, 8w 0.599** — every +one within 1.5 SE of target (`-0.19`, `+0.70`, `+0.17`, `-1.44` SE). Per-bin coverage by record +count is **0.576 / 0.667 / 0.648 / 0.662 / 0.672** for `<=30 / 30-100 / 100-300 / 300-1k / >1k` +upgraded records — no significant count trend. + +**The count effect was mostly F28 in disguise.** Sparse-bin (`<=30` records) coverage runs +**0.422 at 48h blocks → 0.606 at 1h**. A sparse bin at a long block length is short of records *and* +of blocks; shorten the block and the bootstrap recovers. The count term is not (yet) justified: the +block length was. + +**The anticipated short-campaign bias was an artefact of 4 replicates.** The committed compare +baseline (n=4) shows a 1-week bias of ~-0.7 pp, which motivated a bias component. At n=64 the 1-week +bias is **-0.15 pp** against a 1.29 pp spread (bias is 12% of RMS). Removing the bias entirely +*lowers* 1-week coverage (0.557 → 0.510 at 48h), so it is not what limits coverage. The n=4 figure +was noise, and this is the concrete payoff of the replicate count. + +**Two real, smaller effects remain open.** + +1. **The error distribution is platykurtic, not normal** — kurtosis `-0.59 / -0.59 / -0.35 / -0.76` + at 1/2/4/8 weeks, Shapiro p < 0.05 at every length. It is flatter than a Gaussian, so coverage + sits *below* what the scale alone predicts: at 8 weeks `std(z) = 1.03` implies 0.668 under + normality but 0.599 is observed. A 1-sigma coverage target assumes a shape the errors do not have. +2. **8 weeks is the weak end, not 1 week** — coverage 0.56-0.60 at every block length, with + `rms/mean_sigma` = 1.08-1.16. This is the *opposite* of the prior. It is only -1.44 SE, so it is + suggestive rather than established. A campaign-level systematic (something constant within a + campaign, which a within-campaign bootstrap is structurally blind to) would explain both this and + the platykurtosis; the seasonal read is consistent (campaigns starting Jul/Aug are biased + -0.31/-0.35 pp, Dec/Feb +0.09/+0.10 pp) but not established. Adding a floor `f` in quadrature + lands pooled coverage at 0.685 for `f = 0.10 pp`, but the per-length implied floors are + inconsistent (0.52 / 0.00 / 0.00 / 0.16 pp at 1/2/4/8 weeks), so **no floor is proposed on this + evidence** — it would be fitting a term to 1.4 SE of noise. + +**Design implications.** Do not add the count term: at the block length now defaulted to (F28) there +is no count trend left for it to correct. The residual candidates, in the order the evidence supports +them, are: (1) a campaign-level systematic floor, if a larger ensemble confirms the 8-week gap — +which needs replicates, since 64 leaves it at only -1.44 SE; (2) a shape correction, since the +platykurtosis is the most reproducible non-normality here and it caps achievable 1-sigma coverage +even when the scale is right. + +**Method note.** The three profiles are near-redundant for calibration: their signed errors correlate +**0.977-0.995** (they reuse the same `(turbine, treatment_start)` draws and differ only in injected +Cp). 768 headline cells are ~64 independent draws. Quote coverage SE on replicates, never on rows. + +--- + +## F28 — The 48h block prior is wrong for a fast toggle: the *paired* residual's correlation scale is ~1-3h, and a 48h block under-covers (0.622 pooled, 0.557 at 1 week) by starving the bootstrap of blocks + +*2026-07-15 (toggle-specialist uncertainty, round 1). Reproduce: +`uv run python -m benchmarking.baselines.study_toggle_specialist_uncertainty --block-hours 1 2 3 6 12` +plus the default grid; 64 replicates x 3 profiles x 1/2/4/8 weeks. Coverage target 0.683, SE 0.058.* + +`block_hours` defaulted to 48 on the prior that turbine-to-turbine relationships autocorrelate on +roughly that scale. **That prior is about the wrong quantity.** The bootstrap resamples the residual +of the *paired on-vs-off comparison*, and Hill of Towie's toggle alternates every 20 minutes +(`DEFAULT_TOGGLE_PERIOD` = 40 min), so on and off ride the same weather and nearly all the slow +structure — direction, wake state, density, season — cancels between numerator and denominator. + +**Measured autocorrelation of the paired hourly test/reference ratio residual** (8 replicates, +8-week placebo campaigns), mean over replicates: + +| lag | 1h | 3h | 6h | 12h | 24h | 48h | +|---|---|---|---|---|---|---| +| autocorr | 0.181 | 0.065 | 0.047 | 0.034 | 0.032 | **0.003** | + +The correlation scale is **~1-3 hours**. At 48h there is nothing left to capture. + +**Headline coverage is monotone decreasing in block length**, crossing the target at ~2-3h: + +| block | 1h | 2h | 3h | 6h | 12h | 24h | 48h | 96h | +|---|---|---|---|---|---|---|---|---| +| pooled coverage | 0.775 | 0.695 | 0.707 | 0.672 | 0.665 | 0.642 | **0.622** | 0.577 | +| 1-week coverage | 0.766 | 0.719 | 0.724 | 0.672 | 0.667 | 0.620 | **0.557** | 0.438 | + +**Root cause of the long-block failure: circular-block overlap collapse, not noise.** A long block +does not merely make sigma noisy — it biases it **low**. With `n_draw = ceil(T/L)` blocks drawn from +starts anywhere in the campaign, a large `L/T` makes every resample overlap almost completely, so +the resample spread collapses. It is worst where `T/L` is small: at 1 week with 96h blocks +`T/L = 1.75`, mean sigma 0.86 pp against an actual RMS error of 1.30 pp, coverage **0.438**. At 8 +weeks the same 96h block gives `T/L = 14` and sigma falls only ~7% from its 6h value. A synthetic +control (iid noise, 8-week campaign, so `T/L >= 28` throughout) reproduces the expected flat +sigma-vs-L to within ±5%, confirming the collapse is a small-`T/L` effect rather than a bug. + +**Consequence: there is no plateau to read.** Standard practice picks the block length where +sigma-vs-L flattens. Here sigma is flat-to-falling across the whole range, because there is no +autocorrelation to capture and the only remaining gradient is the collapse. **Block length must be +chosen by coverage against truth, not by the sigma curve** — which is why the sweep is scored rather +than plotted alone. + +**Coverage degrades in _both_ directions, so the good region is bounded.** Below ~2h the bootstrap +**over-covers** (1h: headline 0.775, sigma ~1.2x the 12h value). The cause is not established: an +on/off imbalance hypothesis (a 1h block spans only 1.5 cycles of the 40-minute toggle) was **not** +reproduced by synthetic controls with either a constant or a varying reference. So 1h is not adopted +on the strength of its score alone. + +**Applied: `DEFAULT_BLOCK_HOURS` 48 → 6.** The data does **not** distinguish 2h/3h/6h — mean +|coverage - 0.683| across the four reads (headline pooled, headline 1-week, per-bin pooled, per-bin +sparse) is **0.046 / 0.047 / 0.049**. The pick is therefore on robustness, not score: 6h spans 9 +toggle cycles (the most margin from the ~2-cycle edge where the over-coverage sets in), is ~2-6x the +measured correlation scale, still leaves 28 distinct blocks in a 1-week campaign, and stays sane for +a slower toggle where 2h would be a single cycle. **The default is only a default**: block length is +properly a function of the toggle period and campaign length, so a campaign whose toggle period is a +large fraction of 6h must raise it. + +Changing it does not touch any uplift: `study_toggle_methods_compare` still reports +`toggle_specialist: UNCHANGED` (max delta 5e-07 pp), and +`TestUncertaintyDoesNotChangeUplift::test_block_length_moves_sigma_and_nothing_else` asserts the +point estimate is bit-identical across block lengths. + +**Generalisation.** The right block length is set by the toggle period and the campaign length, not +by an absolute number of hours: enough cycles per block that on/off stay balanced, and enough blocks +per campaign that circular overlap does not collapse the spread. A downstream project with a slower +toggle needs a proportionally longer block, so a rule of the form `L = k x toggle_period` (with a +guard on `T/L`) would travel better than any fixed default. Not implemented — it needs its own +evidence. + +--- + +## F27 — `toggle_specialist` can report a per-power-bin uplift without a model, but only with a **per-bin** `rho_base` and a reference-derived bin label; the global-`rho_base` shortcut is ~17 pp wrong + +*2026-07-15 (per-power-bin uplift for the toggle specialist). Reproduce: the estimator comparison is +`TestPerBinIsNotBiasedByTheBaselineRatio` in `tests/benchmarking/baselines/test_toggle_specialist.py`; +the regression baseline is `study_toggle_methods_compare --update-baseline`.* + +`toggle_specialist` now accepts `conditions=("power",)` + `rated_power_kw` and returns a per-power-bin +uplift alongside its headline. Getting there required rejecting two estimators that both look +reasonable, and the F23/F24 reasoning for `power_model` carries over almost unchanged. + +**The estimator.** Bin **both** segments on `rho_base_global * ref_total` — the test turbine's +*predicted untreated* power — then report `rho_up(b) / rho_base(b) - 1` per bin. + +**Why not the global `rho_base`.** Using one global denominator, `u(b) = rho_up(b)/rho_base_global - 1`, +has an attractive property: the per-bin numbers then aggregate back to the headline exactly. It is also +badly wrong. The test-to-reference ratio genuinely varies with power (different turbines, different +wakes, saturation near rated), so the estimator reads that structure as uplift. Measured on a synthetic +case where `k = test/ref_total` falls 0.9 → 0.7 across the power range: **worst per-bin error 16.5 pp on +a placebo, 17.3 pp at a true +5%**. The exact aggregation is bought by smearing the baseline's +power-dependence across the bins. + +**Why not the test turbine's own power as the bin label.** This is the direct analogue of F23/F24, and +its failure mode is instructive: it scores **0.05 pp on a placebo** — i.e. the obvious zero-uplift test +*passes it* — and only breaks once a real uplift exists, at **1.22 pp on a true +5%** (a 24% relative +error). The mechanism: a real uplift shifts the test turbine's power, so the treated rows in a bin +correspond to *lower* untreated power than the baseline rows in it; against a power-dependent `k` that +mismatch becomes bias. **A placebo cannot discriminate these estimators** — the discriminating test is a +*constant non-zero* uplift, which must read that same uplift in every bin. Worth remembering for any +future per-bin work. + +**The chosen estimator scores 0.05 pp in both cases**, and holds under 2% noise. + +**Two deliberate departures from `power_model`'s conditional.** +- **No re-levelling.** Per-bin numbers do not aggregate exactly to the headline, and are not rescaled to. + `power_model` needs its λ because its per-bin shape comes from a `sqrt` of two fits that does not + aggregate; here the per-bin numbers are direct energy ratios, and rescaling them onto an identity that + the per-bin `rho_base` deliberately broke would be a fiction. `sum_actual` / `sum_counterfactual` are + returned so the gap stays inspectable. +- **No imputation.** An uncovered bin reports NaN with `n_records = 0`, not a bfill-then-0-at-rated + prior. This falls out by construction (no baseline rows → no `rho_base(b)` → NaN counterfactual). A + downstream per-bin decision rule wants a sparse bin to land on "keep going"; an imputed value would + manufacture confidence that is not in the data. + +**`power` is the only axis this method will ever support.** It is reference-derived, so the treatment +cannot move a row between bins. Binning by the test turbine's ws/TI would condition on post-treatment +signals, which is exactly the property this method exists to keep (`conditions=("ws",)` raises). + +### Side observations +- **A new regression harness:** `study_toggle_methods_compare.py` scores `toggle_specialist` + + `power_model` on HoT over a placebo and a symmetric +/-2% Cp pair × 1/2/4/8 **weeks**, against a + committed baseline. The per-bin change diffed clean on it: `toggle_specialist`'s 12 cells all moved by + **~1e-9** (i.e. unchanged), confirming the per-bin work left the headline alone. +- **`power_model` is not reproducible run to run, despite `seed` — and this sets a floor on every A/B + against it.** Measured directly (two runs of *identical* code, `--profiles cp_0pct`, separate output + dirs): `power_model`'s bias moved **5.0e-4 (0.05 pp) at campaign_weeks=1**, and **exactly 0.0 at 2 + and 8 weeks**. `toggle_specialist` moved **exactly 0.0 in every cell**. + - **Mechanism:** `seed` governs sampling, not LightGBM's threaded float reduction order. The noise is + **sparsity-driven**, not uniform: with a week of data the model sits near a split boundary, so a + tiny float difference flips a tree and visibly moves the estimate; with 8 weeks the split decisions + are far from the margin and the result is bit-identical. Hence the noise appears *only* at the short + campaigns — the very regime this study exists to probe. + - **This retroactively explains `study_power_model_compare`'s `_MATERIAL_PP = 0.1`**: it is not an + arbitrary comfort band, it sits at ~2x this measured floor. + - **Consequence for the harness:** bands are **per method** (`_UNCHANGED_ATOL`): `toggle_specialist` + 1e-7 (effectively bit-exact), `power_model` 1e-3. A single shared band would either cry wolf on + every power_model re-run or throw away toggle_specialist's much stronger detector. + - **Process note:** the design asserted "an unchanged method must diff to **exactly 0.0**". That was + wrong on two counts (this nondeterminism, plus `record_baseline` storing cells rounded to 8 dp), and + the first real before/after run disproved it. The first correction guessed the floor at 1e-4 from + indirect evidence — also wrong, by 5x, and only caught by running the same code twice. **Measure the + noise floor before setting a band.** +- **`toggle_specialist` has a real short-campaign bias:** at 1 week its bias is ≈ **−0.7 pp** with ~1 pp + spread, consistently across all three profiles (vs ≤0.2 pp at 2+ weeks). Systematic, not noise; it sets + a floor on what a 1-week toggle campaign can resolve. Not investigated here. +- **Provenance flaw in both study scripts:** `_git_commit()` stamps HEAD when the baseline is *written*, + not when the run *started*. On a ~20-minute sweep a commit landing mid-run mislabels the record (the + first toggle baseline stamped `04f36d6-dirty` though it measured `f6b509b`'s method code). Worth + capturing the commit at run start. + +--- + +## F26 — Issue 17: unifying the toggle conditional onto the all-data training window (from campaign-only) blows up the tail bins — the campaign-only restriction is confirmed necessary; both conditional asymmetries are deliberately retained + +*2026-07-14 (Issue 17, scope item 2 — training-window unification). Reproduce: +`study_power_model_compare --modes toggle --profiles cp_0pct ti_dependent_cp ws_dependent_cp` with the toggle +conditional match's `baseline_sel` switched from `campaign_baseline` (strict interleaved off-rows) to +`training_baseline` (pre-campaign + off-rows), vs the committed campaign-only; diffed against the committed +baseline (before = campaign-only, after = all-data). Separate output dir; no baseline change. Prepost is +unaffected by construction (there `campaign_baseline == training_baseline`), so this is a toggle-only test.* + +**Result — a clear regression, worst in the tails.** +- **Overall P50: bit-identical** (the change touches only the conditional match) — correctness check passed. +- **Every conditional axis got worse in both |bias| and spread:** mean |bias| Δ ti **+2.12**, ws **+1.76**, + power **+0.10** pp; mean spread Δ ti +1.57, ws +1.98, power +0.40 pp. +- **The tails explode.** Highest-TI `(0.4,0.45]` went from ~5–9 pp to **60–97 pp** bias; lowest-ws cut-in + `(0,2]`/`(2,4]` swung to **−15…−21 pp**. These are exactly the drift-contaminated bins F15 warned about: + matching temporally-distant pre-campaign rows against campaign on-rows reads reference/era drift as per-bin + uplift. + +**Root cause / why the F17 floor doesn't rescue it (and why decay wouldn't either).** Adding pre-campaign rows +*increases per-bin counts*, so tail bins that were below the F17 count floor — and therefore safely **imputed** — +now clear the floor and are **trusted** with a drift-contaminated two-direction shape. More data makes it worse +by promoting garbage bins from imputed to trusted. Adaptive-half-life decay (the issue's proposed mitigation) +would downweight those rows in the *fit* but they still *count* toward the floor and still populate the matched +pairs, so it would not remove the promotion mechanism. This matches the issue's own note that the genuine fix is +"weighted **or restricted** matching" — and the *restricted* matching is precisely what `campaign_baseline` +already does. + +**Decision: rejected — keep the toggle conditional on `campaign_baseline`.** Combined with F25, **Issue 17 +closes via its second branch:** both conditional/headline asymmetries are *deliberately retained*, each with +fresh post-floor evidence for why — +1. *Conditional match is campaign-only (toggle), not all-data* — because unifying it reads drift as uplift and + the count floor amplifies rather than fixes it (this finding). +2. *Conditional fits are unweighted, headline is decay-weighted* — because recency weighting on the conditional + only churns negligible tail bins (F25). + +The energy-aggregation identity and overall P50 were preserved throughout (both bit-identical in every A/B). +**A better unification direction remains open and is the recommended next step:** F22 showed the *opposite* move — +make the toggle **headline** campaign-only (pull it toward the conditional) — is neutral-to-better on overall P50 +and unifies the data path from the other side; that is genuine design work for a supervised session, not tried +here. + +--- + +## F25 — Issue 17 re-trial: recency-weighting the conditional direction fits (now the F17 count floor exists) is a no-op for toggle and only churns negligible extreme-tail bins for prepost — rejected + +*2026-07-14 (Issue 17, scope item 1 — the specific F14/F16-flagged re-trial). Reproduce: +`study_power_model_compare --modes prepost toggle --profiles cp_0pct ti_dependent_cp ws_dependent_cp` with the +adaptive-half-life time-decay `weights` threaded into `_fit_direction` (the matched forward/reverse conditional +fits), vs the committed unweighted conditional; diffed against the committed baseline (before = unweighted, +after = weighted). Separate output dir; no baseline change.* + +**Context.** F16 held time-decay weights off the conditional direction fits because pre-floor they destabilised +sparse extreme-condition bins (a degenerate tail fit could read three-digit per-bin uplift). Issue 14 deferred +lifting that until the per-bin count floor existed; the floor shipped in F17, so this re-trials the weighting. + +**Result.** +- **Overall P50: bit-identical** in both modes (weighting touches only the conditional fits, not the headline) — + a correctness check that passed. +- **Toggle: every conditional cell bit-identical.** As predicted from the code: the conditional match is + campaign-only, and those rows all sit *inside* the campaign interval where the decay weight is exactly 1, so + weighting is a no-op for toggle. +- **Prepost: change confined to extreme sparse tail bins.** Mean |bias| moved only marginally (ti −0.28, ws + −0.05, power −0.04 pp), but per-bin swings were large (ti −11.6→+1.6, ws −7.4→+2.5 pp) and **entirely in the + tails**: the biggest "improvements" are the highest-TI `(0.4,0.45]` bins dropping from ~54 pp bias to ~42 pp + (both garbage, negligible energy) and the lowest-ws cut-in bins `(0,2]`/`(2,4]` with ±7–16 pp ratio-instability + bias (F5's second failure mode). Meaningful populated bins are essentially unchanged, and **ws spread got + worse** (+0.37 pp mean). + +**Decision: rejected — keep the conditional fits unweighted (F16 stands, now with post-floor evidence).** The +floor removes the three-digit blowups, but weighting still only reshuffles negligible-energy tail bins in both +directions; it buys no accuracy on the bins that carry the energy and costs a little ws spread. Closes Issue 17 +scope item 1: the missing half-life on the conditional path is deliberately retained as an asymmetry, because +adding it does nothing useful. The plumbing (`weights=` on `_fit_direction`/`_estimate_conditional`) was reverted +along with this rejection. + +--- + +## F24 — binning *both* conditional directions on real power (instead of on the prediction) is a wash vs the F23 fix on the tested profiles, and is theoretically less robust — rejected + +*2026-07-14 (Issue 17, follow-up A/B to F23). Reproduce: `study_power_model_compare --modes prepost toggle +--profiles cp_0pct ti_dependent_cp ws_dependent_cp` with the power frame's bin labels temporarily set to +`fwd_cond=y[mu]`, `rev_cond=y[mb]` (both real power) vs the committed F23 default (`pred_up`/`pred_base`); scored +by `conditional_benchmark_comparison` against the accepted baseline, so before = F23 prediction-based, after = +both-real-power. Separate output dir; no baseline change.* + +**Motivation.** With F23 fixed by moving the *reverse* label off its own numerator (`y[mb]`→`pred_base`), an +alternative symmetry is to move the *forward* label onto real power too (`pred_up`→`y[mu]`), i.e. bin both +directions by actual power. A-priori concern: the forward side's real power `y[mu]` is **post-treatment** (design +§3) and is *also the forward ratio's numerator*, so this re-introduces regression-to-the-mean on the forward side +and makes the forward axis shift with the treatment. + +**Result — a wash.** Mean per-bin |bias| change across the power axis was **−0.00 pp in both modes** (prepost: +4 bins better / 5 worse / 9 neutral; toggle: 6 / 4 / 8), with only scattered ±0.3 pp per-bin moves and no +systematic winner. The predicted §3 degradation did **not** appear: on `cp_0pct`/`ti_dependent_cp`/`ws_dependent_cp` +the uplift is 0–small, so treated vs untreated power seldom crosses a 20%-of-rated bin edge and the +post-treatment bin-reassignment smearing is second-order. The one visible tell in the predicted direction: the +placebo's lowest-power bin got *worse* under both-real-power in prepost (−0.25→+0.53 pp), consistent with RTM +leaking back on the forward side. + +**Decision: rejected — keep the F23 prediction-based labelling.** It is empirically no worse here and +theoretically more robust: no RTM on either side, and a treatment-invariant axis. **Caveat / where the difference +should actually show:** the covered profiles are exactly the low/zero-uplift cases; the profiles where §3 +smearing would be largest — `rated_plus_5pct` (uprate) and `cp_plus_10pct` — are not in `COVERED_PROFILES`, so +this A/B cannot see them. The theoretical edge of prediction-based is expected to matter there, not on these +three. Worth a targeted check if a future cycle extends the conditional before/after view to a large-uplift +profile. + +--- + +## F23 — the `power`-axis conditional uplift's "positive at low power, negative at high power" tilt was regression-to-the-mean from binning the reverse direction on its own noisy numerator; labelling both directions by the counterfactual prediction removes it + +*2026-07-14 (Issue 17, power-axis correctness). Reproduce: `benchmarking.baselines.study_power_model_compare` +(`conditional_benchmark_comparison_.csv` + `benchmark_comparison_.csv`); first isolated on the +`cp_0pct` placebo prepost, then confirmed on the full both-mode sweep (7 profiles, campaigns {1,2,3,6,12} mo, +4 replicates, seed 0). One-line change in `method.py:_conditional_by_bin`'s power frame: the reverse-direction +bin label `rev_cond` moved from `y[mb]` to `pred_base`.* + +**Observation.** On `cp_0pct` (true uplift 0 in every bin, so per-bin estimate = pure bias) the `power`-axis +conditional uplift ran monotonically **positive at low power, negative at high** — at 12 mo prepost: `(-230,230]` +**+8.24 pp**, `(230,690]` +2.52, `(690,1150]` +0.77, `(1150,1610]` −0.59, `(1610,2070]` −1.32, `(2070,2530]` +−0.98. A flat-zero truth read as a strong slope; the same shape appeared on the real-uplift profiles and in +toggle. + +**Root cause — binning the reverse ratio on its own numerator.** A condition axis' per-bin shape is +`1+u_b = sqrt((1+r_fwd)/(1+r_rev))` (`_combine_uplift`), which cancels a *common* per-bin multiplicative +shrinkage `s` — valid only when both directions bin a given operating point into the same bin. For ws/TI both +directions read the same treatment-invariant signal, so `s` cancels. The `power` frame instead labelled the +**forward** side by its counterfactual prediction `pred_up` (a regressor) but the **reverse** side by the actual +baseline power `y[mb]` — which is *also the reverse energy ratio's numerator* (`Σy[mb]/Σpred_base`). Binning a +ratio on its own numerator selects each bin on that variable's noise: a low-power bin over-selects downward +noise (`Σy[mb] < Σpred_base` → `r_rev < 0`), a high-power bin over-selects upward noise (`r_rev > 0`). So +`r_rev` tilts negative→positive across power. The forward side, binned on a prediction, carries no matching +tilt, so nothing cancels it; and because `r_rev` sits in the **denominator** of the combine, the tilt inverts: +low power `1+r_rev<1 → shape>1 → +uplift`, high power `shape<1 → −uplift`. It is classic regression to the mean, +the same model-error signature F5 saw in the residual-vs-actual-power diagnostic — surfaced into the estimate +once `power` became a scored conditional axis (F17). + +**Fix.** Label the reverse side by its counterfactual prediction `pred_base` too, so both directions bin on a +(treatment-invariant) *prediction of the same untreated power*: neither side bins on its own noisy numerator (no +RTM), and the per-bin shrinkage is common again so it cancels in the combine — which is exactly what the combine +was designed to assume. The old comment's objection (`pred_base` is "a treated estimate") does not bite: +`pred_base` carries the same shrinkage the forward side does, and cancelling that shrinkage is the whole point of +the two-direction combine. The bin label changed; the ratio contents (`y[mu]/pred_up`, `y[mb]/pred_base`) did not. + +**Evidence.** +- *Placebo, isolated.* `cp_0pct` 12 mo prepost power bins collapsed to truth: `(-230,230]` +8.24→**−0.25 pp**, + `(230,690]` +2.52→+0.09, `(690,1150]` +0.77→+0.18, `(1150,1610]` −0.59→+0.08, `(1610,2070]` −1.32→+0.23, + `(2070,2530]` −0.98→−0.02. Every bin `better`; all six within ±0.25 pp of 0. +- *Full both-mode sweep, benchmark diff (mean per-cell |bias| change vs the pre-fix committed baseline, pp; + Δ<0 = better):* + + | mode | overall | ws | ti | power | + | --- | --- | --- | --- | --- | + | prepost | 0.000 | −0.000 | 0.000 | **−2.177** | + | toggle | −0.000 | −0.000 | −0.000 | **−1.665** | + + The change is fully isolated: overall/ws/ti move ≤0.005 pp (float noise) in both modes, so **overall P50 is + unchanged** and the energy-aggregation identity is preserved. On `power`, 149/210 prepost cells improved by + >0.5 pp, 11 worsened. + +**Residual (not fixed, documented).** The worsening concentrates in one bin, `(1610,2070]` (≈0.7–0.9 of rated — +the power-curve knee), which moved from ~0 to **~+2 pp** across several profiles. This is genuine model +conditional-calibration error at the rated ceiling (asymmetric residuals where predictions clip), previously +*masked* by the larger RTM tilt — not a new artifact. It is a candidate for the F5 baseline-residual calibration +idea; left for a later cycle since the net axis |bias| still dropped ~2 pp. + +**Decision: accepted.** `rev_cond=pred_base` is committed as the default; `study_power_model_compare_baseline.json` +regenerated from the full sweep (both modes) and promoted via `--accept-candidate`. Uncommitted (user does git). + +--- + +## F22 — for toggle, a campaign-only `power_model` *headline* (drop pre-campaign data entirely) is neutral-to-better and leaves the conditional shape unchanged — bears on Issue 17 + +*2026-07-14 — a throwaway A/B while scoping Issue 17 (reconciling the conditional estimator's data +path with the headline). Reproduce: `benchmarking.harness.score_study`, toggle, profiles +`cp_0pct`/`ws_dependent_cp`/`ti_dependent_cp`, campaigns {1, 12} mo, 4 replicates, seed 0; default +`PowerModelMethod` vs the same method wrapped to drop rows before the toggle start first (reusing +`naive_ratio.restrict_to_campaign`). The "default" arm reproduced the committed toggle cells in +`study_power_model_compare_baseline.json` bit-identically, so the A/B is trustworthy.* + +**Where pre-campaign data enters `power_model` at all (toggle):** only the **headline** counterfactual +fit (`baseline_sel` = pre-campaign rows + campaign-off rows, adaptive-half-life weighted, F20). The +**conditional** step already excludes pre-campaign rows (`baseline_sel & _campaign_mask`), so dropping +pre-campaign data only touches the headline. + +**Overall P50** (pp; score = √(bias²+spread²), lower better), mean over the three profiles: + +| campaign | arm | bias | spread | score | +| --- | --- | --- | --- | --- | +| 1 mo | default (all-data) | **−0.31** | 0.23 | 0.39 | +| 1 mo | campaign-only | **+0.02** | 0.34 | 0.34 | +| 12 mo | default (all-data) | +0.07 | **0.18** | 0.19 | +| 12 mo | campaign-only | +0.07 | **0.11** | 0.13 | + +- **1 mo:** dropping pre-campaign data removes a small negative bias (~−0.3 pp → ~0) but raises spread + (0.23→0.34) — the pre-campaign rows stabilise a data-starved short campaign at the cost of pulling + the estimate down. Net score slightly better without them. +- **12 mo:** bias identical; campaign-only has **lower spread** (0.18→0.11) and better score + (0.19→0.13). With a full campaign the extra history buys no bias reduction and only injects + across-replicate variance (each replicate's pre-campaign window differs). + +**Conditional distributions: essentially unchanged** — per-bin |bias| moved ≤ ~0.04 pp (noise) in +every ws/TI bin across all three profiles. Expected from the code: the two-direction CEM fits already +run on campaign-only matched data in both arms, so their *shape* is identical; only the re-level anchor +(the headline) moves, and at 12 mo the headline barely moves. + +**Implication for Issue 17.** Issue 17 frames unification as pushing the *conditional* toward the +headline (all-data + adaptive half-life + weighted/restricted matching). This probe shows the +**opposite direction is also on the table and looks cleaner for toggle**: make the *headline* +campaign-only (matching the conditional), which unifies the data path, is neutral-to-better on overall +P50, and leaves the conditional untouched. It also **qualifies F21's parenthetical** ("the committed +benchmark already shows all-data winning decisively at 3–12 mo"): that was an inference from the F16 +regime map, not a direct all-data-vs-campaign-only-drop A/B — measured directly here, campaign-only +*improves* toggle spread at 12 mo rather than hurting. Caveats: narrow probe (toggle only, 3 profiles, +2 campaign lengths, 4 replicates); prepost is untouched (its baseline *is* the pre-campaign data, so +there is no such knob there). + +--- + +## F21 — prune the historic opt-in knobs: `power_model` now presents only the winning configuration (Issue 16) + +*2026-07-10 — code hygiene, not a methodology change. Every removed knob lost its A/B and was off/absent +in the shipped default, so the committed benchmark is unchanged (the acceptance test: a default-config +`study_power_model_compare.py` sweep reproduces `study_power_model_compare_baseline.json` bit-identically, +no `--update-baseline`). The point is a smaller surface, less overfit temptation, and code that reads as +the successful approach.* + +### Removed (each off/absent in the default; the findings verdict that retired it in brackets) +- `calibrate_slope` [F14], `calibrate_residuals` [F15] — headline calibrations that never transferred; + with them go `fitting.py`'s `fit_calibration_line` / `cell_residual_calibration` / `CalibrationLine` / + `early_stopped_n_estimators` and the method's `_oof_baseline_predictions` / `_fit_calibration` / + `_residual_corrections`. +- `early_stopping` [F14 — neutral, +15% runtime], `n_seed_ensemble` [F14 — spread is weather-sampling, + not seed noise, +75% runtime]. +- `toggle_estimator="double_ratio"` **and** the temporary `rho_off_scope` knob [F16/F19 — never beats the + counterfactual on score]. With `double_ratio` gone, `toggle_estimator` was single-valued → dropped; the + counterfactual energy ratio is the sole toggle headline (`_fit_predict_double_ratio` / `_rho_off_mask` + removed). +- `time_features` (+ `latitude`, `longitude`) [F11 — all rejected], `era5_derivations` (+ `hub_height_m`) + [F10 — all rejected *as model features*]. `era5_derived.py` and `time_features.py` **stay as utilities** + (CEM matching / `inspect_era5_matching_importance.py` use the derivations); only the model-feature wiring + and the method-surface knobs went. +- The injectable **`model_factory` seam** and its `OUTCOME_MODEL_FACTORIES` (`hgb`/`linear`) registry — + removed entirely: **no driver, example or inspection script used it** (the issue's "if nothing uses it, + remove it too"). LightGBM `make_outcome_model` is the sole outcome model; a future Phase-2 learner + re-introduces a seam when a driver actually needs one. `fitting.py` thinned to just `time_block_folds`. + +### Kept (so it isn't re-litigated) +Settled defaults: `era5_exclude=CURATED_ERA5_EXCLUDE` (F13), `availability_feature=False` (F13), +`reference_stat_cols` (schema), `matching_vars`/`matching_bin_edges` (F6), the adaptive time-decay default +and its `time_decay_half_life_days` expert override (F20), `_MIN_BIN_MATCHED_COUNT` conditional floor (F17). + +### `toggle_campaign_only` — confirmed redundant, removed from `power_model` +The F20 adaptive half-life is a *soft* campaign-only at short campaigns, so the hard knob was expected +redundant. One cheap A/B settled it (`inspect_short_campaigns.py`, toggle 1–2 mo, placebo `cp_0pct` + +recovery `cp_plus_3pct`, 4 replicates): adaptive default (`tco=False`) vs adaptive + `tco=True`, overall +score pp: + +| profile | mo | default | tco_true | +| --- | --- | --- | --- | +| cp_0pct | 1 | 0.380 | 0.334 | +| cp_0pct | 2 | 0.201 | 0.191 | +| cp_plus_3pct | 1 | 0.391 | 0.343 | +| cp_plus_3pct | 2 | 0.206 | 0.194 | + +Campaign-only is marginally better at 1–2 mo (≤0.05 pp, near-noise on 4 replicates; the default carries a +small −0.3 pp bias, campaign-only near-zero bias but higher spread), and the committed benchmark already +shows all-data winning decisively at 3–12 mo — where a hard `tco=True` would hurt. The soft adaptive +window captures the short-campaign benefit without the long-campaign cost, so the knob does not earn its +place: **removed from `PowerModelMethod`** (always all-data; the conditional step still matches within the +campaign via `_campaign_mask`). `naive_ratio` keeps its own `toggle_campaign_only` — a different method, +untouched. + +### Mechanics +Deleted the knobs, their config-validation branches, their `_config_params` entries and plumbing, and +their unit tests; thinned `fitting.py`; the run-config YAML loses the removed keys (a per-run diagnostic, +not the committed benchmark). `inspect_short_campaigns.py` lost its `double_ratio*`/`tco_true` arms; +`study_power_model_compare.py` help/docstring examples updated off the removed `era5_derivations`. Public +`PowerModelMethod` surface dropped from 30 to 21 constructor params. `poe all-fast` green. Acceptance test +passed: the full default-config sweep is bit-identical to the committed benchmark — **0 better / 0 worse of +819 prepost + 791 toggle conditional cells** on spread, score and |bias|, leaderboard deltas all zero. + +--- + +## F20 — the headline training window self-configures: an adaptive time-decay half-life (2 × campaign duration) is the new default, subsuming the fixed 548 d; big short-campaign wins, tied long, in both modes (Issue 15 deliverable 2) + +*2026-07-10 — Issue 15's mechanism-justified fallback after F19 ruled out the `double_ratio` path. New +`PowerModelMethod` default `adaptive_time_decay: bool = True`; the headline fit's time-decay half-life +is now `_TIME_DECAY_CAMPAIGN_MULTIPLE (=2.0) × campaign_duration_days` via a new +`_effective_half_life`. `time_decay_half_life_days` is demoted to an **expert override** (default flipped +548.0 → `None`, used only when `adaptive_time_decay=False`; a guard forbids setting both). Conditional +two-direction fits stay unweighted (F16). A/B'd via `study_power_model_compare.py` (plain run = +adaptive-default vs the fixed-548 committed benchmark), placebo `cp_0pct` both modes for the coarse `k` +sweep, then the full 7-profile both-mode sweep for the ship.* + +### Mechanism — a scale-free "trust window", not a lookup +The best fixed half-life is regime-dependent (F16: ~90 d helps short campaigns in **both** modes; ≥1 yr +is safe long; 548 d fixed was a compromise). Making the half-life **proportional to the campaign's own +duration** — `k × campaign_days` — gives a short half-life for a short campaign (down-weight the stale +pre-campaign era that dominates a sliver campaign → less bias *and*, in toggle, less spread) and a long +one for a long campaign (use the plentiful recent data). One dimensionless `k` = "trust pre-campaign +data within ~k campaign-durations". A scale-free multiple is chosen over a length→half-life lookup +precisely to avoid benchmark overfitting: it is mechanism-anchored and generalises to any campaign +length or dataset span, including F16's "20 yr of SCADA, 1-month campaign" case by construction. +`k=2` → 1mo≈60 d, 3mo≈182 d, 12mo≈730 d (≈ today's 548 d regime at 12 months, so a strict +generalisation of the accepted default). + +### Coarse `k` sweep (placebo `cp_0pct`, both modes, overall score pp; all vs fixed-548 benchmark) +All three adaptive `k` beat fixed-548 on pooled mean in **both** modes (prepost 0.55 → ~0.50, toggle +0.34 → ~0.25). Per-length the best `k` is scattered across {1.5, 2, 3} — the signature of a flat region +where finer tuning would fit placebo noise. **k=2 chosen**: best prepost pooled mean (0.493), 2nd toggle +(0.248, a hair behind k1.5's 0.232), and it wins the headline **toggle-1mo** case (0.380 vs k1.5 0.400, +k3 0.482). It is the middle of the bracket (least overfit-prone) and matches the a-priori mechanism +anchor; chasing k1.5's toggle-2mo edge would cost prepost-1mo/6mo. `k` lives as the module constant +`_TIME_DECAY_CAMPAIGN_MULTIPLE`, **not** a user knob, so it is documented without inviting per-dataset +tuning. + +### Full 7-profile sweep verdict (overall P50, pooled per length, dScore pp; <0 = adaptive better) +| mo | prepost Δ | toggle Δ | cells worse >0.1pp | +| --- | --- | --- | --- | +| 1 | +0.067 | **−0.368** (all 7) | 0 | +| 2 | **−0.338** (all 7) | **−0.144** (all 7) | 0 | +| 3 | +0.014 | +0.024 | 0 | +| 6 | −0.027 | −0.005 | 0 | +| 12 | −0.005 | +0.003 | 0 | + +Pooled mean score **prepost 0.559 → 0.501, toggle 0.350 → 0.252**. **Zero cells regress by >0.1 pp** +anywhere; the only positive deltas (prepost-1mo +0.067, 3mo +0.01–0.02) are uniform across profiles and +inside the 0.1 pp materiality band — i.e. tied. The prepost-1mo +0.067 is a spread effect (the shortest +prepost campaign wants maximal baseline; a 60 d half-life thins it) and neither a longer `k=3` (+0.084) +nor shorter `k=1.5` (+0.114) helps it — an inherent short-prepost tension the adaptive rule accepts to +buy the large toggle short-campaign gains. Conditional decomposition (re-levels to the now-adaptive +headline; the two-direction shape is unweighted and unchanged) is neutral-to-better: prepost mean +|bias| −0.87 pp (46 better/17 worse of 69), toggle −0.10 pp (32/32), no cell flagged a material +regression. Benchmark JSON regenerated on the new default. + +### Decisions / follow-ups +- `adaptive_time_decay=True` is the shipped default; the fixed half-life is the expert escape hatch. +- **`toggle_campaign_only` demotion not yet decided** — the adaptive half-life is a *soft* campaign-only + at short campaigns, so whether `toggle_campaign_only` is now redundant is a separate confirmation + (tracked, not done here). +- The temporary `double_ratio` `rho_off_scope` knob (F19) is still present; removing it (shipping + era-local as `double_ratio`'s behaviour) is the remaining knob-cleanup deliverable. +- `inspect_short_campaigns.py`'s `hl90`/`hl365` arms updated to set `adaptive_time_decay=False`. + +--- + +## F19 — the era-local `double_ratio` gate: no `double_ratio` variant beats the counterfactual default, so the self-configuring toggle default is not an estimator swap (Issue 15 deliverable 1) + +*2026-07-10 — Issue 15's first, gated step. New temporary A/B knob `PowerModelMethod.rho_off_scope` +(`"campaign"` default / `"all"`): `double_ratio`'s calibration ratio `rho_off = Σy_off/Σpred_off` is now +measured over the **campaign-window off rows only** (era-local), the fold models still training on all +off rows; `"all"` reproduces the pre-Issue-15 behaviour. New `_rho_off_mask` + threaded campaign mask; +to be removed with the knob cleanup. A/B'd on the toggle `cp_0pct` placebo at 1/2/3/6/12 months via +`study_power_model_compare.py --method-overrides`.* + +### Result — era-local fixes the 1-month bias but does not dominate +Toggle `cp_0pct` placebo, score = √(bias²+spread²) pp (cf = counterfactual default = the committed +benchmark; gl = global-`rho_off` DR; el = era-local `rho_off` DR): + +| mo | bias cf/gl/el | spread cf/gl/el | score cf/gl/el | +| --- | --- | --- | --- | +| 1 | −0.56 / −0.71 / **+0.04** | 0.49 / 0.69 / 0.81 | **0.74** / 0.99 / 0.81 | +| 2 | −0.17 / −0.24 / −0.22 | 0.30 / 0.50 / **0.25** | 0.34 / 0.56 / **0.34** | +| 3 | +0.11 / **−0.01** / +0.41 | 0.20 / 0.31 / 0.34 | **0.23** / 0.31 / 0.53 | +| 6 | +0.15 / **+0.07** / +0.16 | 0.15 / 0.22 / 0.26 | **0.22** / 0.24 / 0.31 | +| 12 | +0.08 / −0.03 / −0.01 | 0.17 / 0.24 / **0.11** | 0.19 / 0.25 / **0.11** | + +Era-local fixes `double_ratio`'s 1-month **bias** (−0.71 → +0.04) by keeping `rho_off` in the ON +window's era, but the campaign-window off set is tiny (**2150 of 101532 rows at 1mo**), so `rho_off` +becomes a noisy ±0.5 % multiplier (1.004/1.009/0.994 across replicates vs the stable global 0.99934) +and **spread balloons**. On the composite score it does **not** dominate: the plain counterfactual +default is best-or-tied at 1/2/3/6 months; era-local only wins at 12mo; global-`rho_off` DR is dominated +almost everywhere. Prepost is untouched (the overrides are toggle-only). Verified on current code, not +just F16's recorded numbers. + +### Root cause — a well-calibrated model leaves `double_ratio` nothing to do +`double_ratio` = `rho_on/rho_off − 1`; its value is cancelling model *miscalibration* shared between +the on/off sides. But with **all-data training the model is already well-calibrated** (logged +`rho_off = 0.99934` ≈ 1), so the ratio-of-ratios only **injects estimation noise** — visible in the +spreads (global-DR spread exceeds the plain counterfactual sum's at every length ≥2mo). `double_ratio` +only earns its keep when the model *is* miscalibrated — i.e. campaign-only / small fits — but F16 found +campaign-only `double_ratio` overcorrects and its 3-month spread balloons. It is boxed between "nothing +to correct" (all-data) and "too noisy to correct" (campaign-only); a smoother `rho_off` does not free it. + +### Decision +No `double_ratio` variant beats the counterfactual default on score → the self-configuring toggle +default is **not** an estimator swap. The mechanism-justified lever is the training window +(drift-vs-shrinkage; F16's crossover), pursued as the adaptive time-decay half-life (**F20**). Era-local +`rho_off` is kept as `double_ratio`'s behaviour (a strict fix to the opt-in); the `rho_off_scope` knob is +temporary and slated for removal in the knob-cleanup deliverable. + +--- + +## F18 — the frozen reference dir regenerated on current code and verified; the compare default repointed at it; a clean four-method bias/spread read (F17 stale-reference follow-up) + +*2026-07-09 — the F17 stale-reference flag closed out. A full overnight run +(`study_overnight_prepost` + `study_overnight_toggle`, both `include_v0=True`, seed 0, 4 replicates, +1/2/3/6/12 months, all seven `overnight_profiles`) was produced on current committed code as +`~/temp/wind-up-benchmarking/badass overnight 20260708/`. `study_power_model_compare.py` now defaults +`_DEFAULT_REFERENCE_DIR` to it (was the unreproducible "30 June" run) and its `_load_reference_methods` +tolerates the timestamped `//` subdir that `start_overnight_run` writes, so an +overnight run drops in as a reference with no manual flattening.* + +### Consistency verified (the run is safe to use as the frozen reference) +- **Config is identical** to the compare grid and the committed benchmark: `campaign_months=[1,2,3,6,12]`, + `n_replicates=4`, `seed=0`, all seven profiles, both modes. +- **v0 now spans the whole grid.** `v0_binned` is present at all of 1/2/3/6/12 months × 4 replicates in + both modes — fuller than the 30-June reference, which had no v0 below 3 months. The compare script + reuses only v0 (`REUSED_METHODS = ["v0_binned"]`), so this is exactly what it needs. +- **power_model matches the committed benchmark within noise.** Reproducing `power_model_leaderboard` + from the run and diffing against `study_power_model_compare_baseline.json`: **toggle** bit-identical + (max |Δ| < 1e-6 on bias/spread/score), **prepost** max |Δbias| = 0.019 pp — well under the 0.1 pp + materiality band. (The reference run's own power_model carries only `overall`+`ws` conditional cells, + no `ti`, because the overnight driver `example_prepost_study.py` omits `wind_speed_sd_col`; irrelevant + to the compare workflow, which recomputes power_model fresh with TI and only reuses v0.) +- Ground truth is method-independent, shared across all four methods per case, so the compare script's + `_check_alignment` guard passes on the full intersection. + +### The four-method P50 read (overall condition, pooled 7 profiles × 5 lengths × 4 reps; pp) +`oracle` = 0.00 bias / 0.00 spread in both modes, confirming the truth anchor. + +| method | prepost bias | prepost spread | toggle bias | toggle spread | +|---|---|---|---|---| +| naive_ratio | −1.02 | 1.78 | +0.14 | 0.44 | +| v0_binned | −0.96 | 1.01 | −0.06 | 0.37 | +| power_model | +0.04 | 0.64 | −0.08 | 0.40 | + +- **Prepost is the hard regime; toggle is easy.** Every method is several-fold tighter and less biased in + toggle. power_model dominates prepost on both axes; in toggle v0 (0.37) and power_model (0.40) are + comparable and naive (0.44) is only slightly behind. +- **naive and v0 under-report by ~1 pp in prepost**, and that negative bias is roughly *constant across + upgrade size* (naive −0.8…−1.3 pp, v0 −0.5…−1.4 pp over all seven profiles) — a floor/detrend offset, + not a scaling error. **power_model removes it** (+0.04 pp, uniform across every upgrade incl. the + `cp_0pct` placebo) — the key robustness result. +- **naive's toggle bias scales with the true uplift** (`cp_plus_10pct` +0.44, `cp_minus_10pct` −0.27 pp): + a proportional over-read the ratio estimator has and the two models do not. +- **`rated_plus_5pct` is v0's consistent weak spot** — its largest |bias| in both modes (prepost −1.44, + toggle −0.69 pp). +- **Precision improves with campaign length** for all methods, most steeply for power_model in prepost + (1 mo 1.00 → 3 mo 0.23 → 6 mo 0.20 pp); at 12 months toggle spreads fall to ~0.14–0.17 pp across the + board. The 1-month prepost cell is where everyone struggles (naive 1.76, v0 1.06, power_model 1.00 pp). + +### Implications +- v0 comparison numbers are now trustworthy (F17's blocker cleared); the leaderboard's v0-vs-power_model + gap in prepost is a genuine current-code result, not a stale-reference artefact. +- Any future overnight run is a drop-in reference (loader handles the timestamp subdir); pointing + `--reference-dir` at a run holding two-or-more run subdirs per mode fails loudly rather than guessing. + +--- + +## F17 — conditional decomposition hardened (count floor + physics imputation + corrected re-level); out-of-the-box method promoted onto the class and scored 1–12 months; the frozen reference is stale (Issue 14) + +*2026-07-08 — Issue 14, both efforts. New module +`benchmarking/baselines/power_model/conditional.py` (`impute_uncovered_bins`, `relevel_conditional`). +New `PowerModelMethod` internals: the per-reporting-bin count floor `_MIN_BIN_MATCHED_COUNT = 50`, +imputation + corrected re-level wired into `_conditional_by_bin`, and a `covered` flag on the per-run +`conditional/` CSV. Class defaults promoted (see below). `study_power_model_compare.py` now scores +1/2/3/6/12 months in both modes with `naive_ratio` recomputed fresh; benchmark JSON regenerated on the +new grid under the new conditional default. A/B'd on the covered profiles (`cp_0pct`, +`ti_dependent_cp`, `ws_dependent_cp`) against the pure-B benchmark; full 7-profile regen for the ship.* + +### Effort B — the benchmarked config is now the out-of-the-box config, scored 1–12 months +- **Defaults promoted onto the class** so a bare `PowerModelMethod` *is* the benchmarked method: + `min_child_samples=50` (F14, merged under user `model_params` on the LightGBM path), + `availability_feature=False` (F13), and `era5_exclude=CURATED_ERA5_EXCLUDE` (F13) with the **untouched + default** applied drop-if-present (a non-Open-Meteo frame is not broken; an explicitly-set exclude + keeps the strict typo guard). `reference_stat_cols` stays driver-level — it names a source-specific + SCADA tag (`wtc_ActPower_min`), i.e. schema description, not tuning. +- **Grid extended to 1/2/3/6/12 months both modes**; `naive_ratio` recomputed fresh (cheap, no wind_up + pipeline) so the merge no longer depends on the reference run's naive; v0 stays reference-only and is + simply absent below 3 months; the alignment guard now checks truth on the fresh∩reference + intersection only. +- **Out-of-the-box power_model, overall P50, mean over 7 profiles [pp]** (regenerated benchmark). The + short-campaign regime (F16) is now visible in the committed benchmark: + | months | prepost bias / spread / score | toggle bias / spread / score | + | --- | --- | --- | + | 1 | −0.56 / 1.00 / 1.15 | −0.57 / 0.50 / 0.76 | + | 2 | +0.04 / 0.53 / 0.54 | −0.17 / 0.30 / 0.35 | + | 3 | +0.14 / 0.23 / 0.27 | +0.11 / 0.21 / 0.23 | + | 6 | +0.44 / 0.20 / 0.49 | +0.15 / 0.16 / 0.22 | + | 12 | +0.12 / 0.34 / 0.36 | +0.08 / 0.17 / 0.19 | + +### Effort A — the conditional decomposition, hardened +- **Every per-bin number is now a trustworthy measured value or a flagged, physics-informed + imputation.** A bin is `covered` only if its two-direction shape is finite **and** both directions + have `≥ _MIN_BIN_MATCHED_COUNT` matched rows; otherwise it is imputed (ws: bfill from the nearest + covered bin above, then 0 uplift above the last covered bin — 0-at-rated; ti: the overall uplift) and + flagged `covered=False`. Never a bare NaN, so `summarize_errors` (which drops non-finite errors) + cannot be gamed by abstention, and the imputation prior is itself benchmarked. +- **Corrected re-level (the pure-bug fix).** `relevel_conditional` pins imputed bins at their imputed + uplift and solves one λ over the **measured** bins only + (`λ = S_m / (Σactual/one_plus_overall − C_i)`), so measured + imputed together energy-aggregate to the + headline **exactly** even when coverage is imperfect (the old re-level solved λ over covered bins only + and silently absorbed the uncovered bins' MWh). Guards: no measured bins or a non-positive denominator + → overall uplift in every bin. +- **Result (A/B vs the pure-B benchmark, 309/321 conditional cells over all lengths + covered + profiles):** overall P50 **bit-identical** (Δbias/Δspread/Δscore = 0.000 pp — the headline is the + single full-window fit, untouched by construction); conditional mean **Δscore −3.7 pp prepost / −2.6 pp + toggle** (Δbias −2.0/−1.4, Δspread −2.7/−2.0). The before/after view (longest campaign, covered + profiles) reads **12 better / 57 ~ / 0 worse** prepost and **15 better / 54 ~ / 0 worse** toggle — the + F7/F9 sparse-extreme "worse" bins are gone. +- **Floor value chosen on evidence, not the one bad bin** (`floor_threshold_evidence.py`, 14.6k bins + from the A/B run): the raw two-direction shape `|u_b|` p90 by per-side matched count is 13 pp (<10, + mostly degenerate → imputed anyway), **60 pp (10–25), 55 pp (25–50)**, 52 pp (50–100), then falls to + 25 pp (100–200), 13 pp (200–500), 10 pp (500+). The combine is untrustworthy below ~50 (a floor of 25 + would leave the wild 25–50 bucket unfloored); 50 is the smallest value that catches it, and raising + higher would impute away ~940 moderately-populated 50–100 bins for no done-when gain. Kish ESS was not + needed (no balance reweighting shipped — see below), so the floor compares raw per-side counts. + +### Decisions +- **Coverage stays method-internal (user decision, trims the issue's "coverage in the leaderboard" + bullet).** The `covered` flag lives only on the per-run `conditional/` CSV and drives the re-level; + the harness seam (`MethodOutput.p50_by_condition`) is unchanged `[condition, condition_bin, + p50_uplift]`. Rationale: once imputation makes every bin finite, `summarize_errors` drops nothing, so + the F16 "deltas dominated by a few exploding cells" problem — the reason a coverage view was wanted — + is dissolved by the imputation itself. The method reports its single best per-bin estimate; the + leaderboard scores that. +- **Per-bin balance (post-stratify each reporting bin to the intersection of the two directions' + ERA5-cell supports) — DEFERRED, not shipped.** The floor + imputation + corrected re-level already + clear every Issue-14 done-when with margin (0 worse bins, conditional score materially better, overall + unchanged), so per the adopt-only-if-it-helps protocol the extra within-bin reweighting is not built — + it would thread ERA5 cell codes and a custom per-bin reduction into the core conditional path for a + residual imbalance the floor already tames (YAGNI). Tracked as a follow-up; the design is a pure + intersect-cell-support mask over the matched fwd/rev rows within each reporting bin, with the floor + switching to Kish ESS `(Σw)²/Σw²` if it is ever adopted. + +### The frozen reference dir is stale (flagged for regeneration) +- The naive-consistency check (`naive_consistency_check.py`, requested to run once before trusting the + reference) **failed**, but in the *good* direction: current-code `naive_ratio` on the `cp_0pct` + placebo has a **much tighter prepost spread** than the frozen 30-June reference (3-mo score 1.91 vs + 7.19, spread 1.57 vs 7.15; 6/12-mo better too), toggle essentially identical. Every input to naive — + `naive_ratio.py`, `filtering.py`, the study config, the turbine set, the data span, and (via the + passing alignment guard) the ground truth — is byte-identical between the reference era and HEAD, and + the reference predates power_model (methods `[naive, oracle, rlearner, v0]`, ~26–30 June), so the + 30-June run was produced by a **local/uncommitted state** git can't reproduce. The committed benchmark + is power_model-only and alignment-guarded, so this does not affect it; naive is now recomputed fresh + (the correct, current-code bar). **The frozen v0 in that reference is from the same stale state** and + should be regenerated (run `study_overnight_prepost` / `study_overnight_toggle`, both `include_v0=True`, + on current committed code) before v0 comparison numbers are trusted. +- **Resolved in F18** — the reference was regenerated on current code and verified consistent; the + compare script now defaults to it. + +--- + +## F16 — a finite time-decay half-life (548 d) is the default, applied to the headline fit only; the double-ratio toggle estimator validated as opt-in; the 1–2-month regime flips the training-window verdict (Issue 13 extension) + +*2026-07-04/05 — three user-directed follow-ups to F15, A/B'd against the post-F15 benchmark. +New: `toggle_estimator="double_ratio"` on `PowerModelMethod` (the naive-adoption hybrid) and +`benchmarking/baselines/inspect_short_campaigns.py` (1–2-month campaigns are outside the committed +benchmark grid; oracle + naive anchor them — the oracle scores exactly 0 there, so the harness +itself is sound at those lengths).* + +### ACCEPTED — `time_decay_half_life_days = 548` (1.5 years), headline fit only; benchmark regenerated +- **Why finite at all:** the method must work on any dataset; with 20 years of SCADA an unbounded + training window would let ancient eras dominate the campaign era. `0.5^(days outside the + campaign interval / 548)`: rows in the campaign weigh exactly 1, year-old data ~63%, + decade-old ~1%. +- **Dose curve (cp_0pct, both modes):** 90/180/365/548/1096 days all leave the toggle overall + neutral-or-better; prepost overall bias nudges up slightly (+0.1–0.2 pp at 3 months) and + everything ≥365 d is indistinguishable on this 2.5-year dataset — the specific choice of 548 + rests on the bounding argument, not a HoT win. Short half-lives (90 d) measurably help + 1–3-month campaigns in **both** modes (prepost 2-month score 0.21 vs 0.59 unweighted; toggle + best-or-near-best at 1/2/3 months) at a small long-prepost bias cost — the documented + short-campaign tuning. +- **Headline fit only.** Weighting the conditional two-direction fits destabilises the sparse + extreme-TI tail bins (a degenerate matched fit in one replicate read +1039 % in one bin; it + appeared at hl365/548 and not at 180/1096 — replicate chance, the F7/F8 tail fragility that + Issue 14's count floor addresses). The matched contrast is already era-insensitive (its common + shrinkage cancels), so the weights buy nothing there. With weights confined to the headline + path the full sweep is clean: prepost conditional ALL Δscore +5.4 (weighted) → **+0.03** + (headline-only); every overall row in both modes inside the ±0.1 pp neutrality band, toggle + slightly better everywhere. + +### Validated opt-in — `toggle_estimator="double_ratio"` (the naive-adoption hybrid) +- The model is demoted to removing condition unfairness; the headline is a ratio of ratios, + ``(Σy_on/Σpred_on) / (Σy_off/Σpred_off) − 1``. Off rows are predicted out-of-fold and on rows + by the same fold-model ensemble (basis-consistent, the F15 rule), so the model's shrinkage / + level bias cancels between the interleaved sides instead of needing to be zero. +- **On the benchmark grid it does exactly what it promises:** toggle placebo headline bias ≈ 0 at + every campaign length (−0.00/+0.08/−0.06/−0.05 pp vs +0.12/+0.16/+0.08/+0.08 for the default), + score neutral. The logged ρ_off (0.9995–1.0006 on ~100k-row fits) is precisely the residual OOF + shrinkage, cancelled by construction. +- **Not the default, for two measured reasons:** (i) with campaign-only training it overcorrects + and the 3-month spread balloons (0.26 → 0.80 — five fold models on ~6k rows are noise); (ii) at + 1–2-month campaigns it is *worse* than the default (score 1.05/0.59 vs 0.90/0.38): with + all-data training, ρ_off is measured across two years of eras while the ON window is a + one-month sliver, so the "cancellation" subtracts the wrong era's miscalibration. Right tool + for ≥3-month toggle campaigns where headline-bias purity matters. + +### The 1–2-month regime check (the reason the user asked for it) +- **The Issue 13 training-window verdict flips below ~3 months:** campaign-only training wins at + 1–2 months (toggle 1-month score 0.334 vs 0.895 for all-data; bias +0.02 vs −0.64) because the + campaign is <5 % of the all-data training set and drift dominates; all-data wins at ≥3 months + where shrinkage dominates. The decay weights interpolate between the regimes: hl90 is + best-or-near-best at 1, 2 *and* 3 months. A campaign-length-adaptive training window (or + half-life) is the natural future refinement — noted for the later-work list, not implemented. +- The 3–12-month verdicts (the committed benchmark grid) all stand: mcs=50, tco=False, and the + F12/F13 feature set were re-checked where they can flip and none reversed on that grid. +- power_model beats the naive anchor at 1–2-month toggle only in its campaign-only configuration + (0.334 vs 0.518 at 1 month); in the default all-data configuration naive wins there — worth + remembering when quoting short-campaign capability. + +### Method notes +- The weights default surfaced a seam conflict: sklearn ``Pipeline`` factories take no + ``sample_weight``. ``_fit_kwargs`` now probes support (``has_fit_parameter``) and fits + unweighted with a warning instead of crashing the factory seam. +- 1–2-month campaigns are deliberately **not** added to the committed benchmark grid (the frozen + v0/naive reference runs don't cover them); `inspect_short_campaigns.py` is the reproducible + driver for that regime. + +--- + +## F15 — residual calibration rejected with a sharpened OOF-transfer rule; `toggle_campaign_only=False` accepted (all-data headline training + campaign-restricted conditional); time-decay weights validated as opt-in (Issue 13) + +*2026-07-04 — Issue 13 verdicts. New `PowerModelMethod` knobs (default off): `calibrate_residuals` +(ERA5-cell residual calibration: full-baseline out-of-fold predictions via the shared time-blocked +folds, mean residual per F6 CEM cell — `matching.cell_codes` is now public — read out under the +upgraded window's occupancy) and `time_decay_half_life_days` (campaign-proximity sample weights +`0.5^(days outside the campaign interval / half-life)`: interleaved campaign rows weigh exactly 1, +pre-campaign rows decay; threaded through every fit). One structural change: with +`toggle_campaign_only=False` the conditional two-direction step now **matches within the campaign +only** — pre-campaign rows serve only the headline fit's training data. A/B'd on the placebo +(`cp_0pct`) against the post-F14 benchmark; acceptance via a full 7-profile sweep. Entering +Issue 13, the prepost placebo headline bias was already −0.02/+0.35/+0.05 pp at 3/6/12 months +(the F3-era uniform −0.4 pp is gone since F13/F14) and toggle +0.36 pp at 3 months (F14: small-fit +tree shrinkage).* + +### REJECTED — ERA5-cell residual calibration, twice; the F14 OOF-transfer rule now covers *shape* +- **v1 (raw cell means)**: prepost bias up nearly uniformly (+0.32/+0.21/+0.13 pp) — the global OOF + residual level (≈ −1 kW: fold models fit on 80% of the rows over-predict relative to the final + 100% fit) leaks into every cell mean and swamps the mix-shift signal. Toggle: right direction but + a short-campaign spread cost (3-month Δspread +0.72 — the F7 failure mode; the level term is + noise at small n). +- **v2 (centred: `cell_mean − global_mean`, the pure mix-shift differential; unseen cells → 0)**: + prepost deltas nearly identical to v1 (+0.33/+0.20/+0.15). The logged corrections under the + upgraded mix are systematically *negative* (−0.05…−4.9 kW) while the observed placebo bias is + *positive* — **the estimated correction has the wrong sign vs the actual bias**. The fold + models' conditional residual *structure* differs from the final refit model's, not just its + level. Toggle adds a sparse-cell artifact: ~450 F6 cells over ~6k off rows ≈ 13 rows/cell of + noise (uniform positive bias shifts decaying ~1/n). +- **The sharpened rule (extends the F14 method note): out-of-fold residuals cannot be transferred + to a refit model — neither their level nor their conditional shape.** Any future residual + calibration must be basis-consistent: estimate corrections for the model actually deployed + (e.g. predict with the fold ensemble itself, or calibrate on a held-out era the final model + never saw). +- **Corollary**: the remaining prepost 6-month +0.35 pp placebo bias is measurably *not* an + ERA5-weather-mix-shift effect (the purpose-built correction moves it the wrong way). 3- and + 12-month prepost already meet the issue's ≲0.1–0.2 pp target; the 6-month anomaly stays open + (candidate mechanism: a seasonal/drift interaction specific to half-year windows). + +### ACCEPTED — `toggle_campaign_only=False` (all-data headline training; benchmark regenerated) +- **Where the all-data damage actually lives.** Every naive all-data arm wrecked the toggle + conditional identically (3-month ti Δscore ≈ +14) regardless of decay weights or calibration — + because the drift enters through the conditional CEM **matching** (pre-campaign rows matched + against campaign on-rows at full weight in the two-direction contrast), not through the fits. + Weighting cannot fix a matching problem; the structural fix (conditional matches within the + campaign only, where its shared-distribution assumption holds) resolves it completely — the + 3-month conditional comes out *better* than the campaign-only benchmark (ti Δscore −1.7). +- **Full-sweep verdict (7 profiles)**: prepost bit-identical (the flip is toggle-only; 0.0 across + all 469 cells). Toggle: 3-month headline bias +0.358 → **+0.121** (Δscore −0.070), pooled ALL + Δscore −0.144, |bias| cells 144 better / 86 worse. Accepted cost: mild spread at 9/12 months + (overall Δscore +0.11/+0.08) — the extra rows only add drift variance once the campaign is + data-rich. Three-method picture: power_model now ties v0/naive at 3-month toggle (0.294 vs + 0.296/0.297), closing its last deficit vs v0 (F4); naive keeps the 9/12-month toggle lead. +- `naive_ratio` keeps campaign-only, per the issue — there the restriction *is* the method's + distribution matching. + +### Validated opt-in — `time_decay_half_life_days` (campaign-proximity training weights) +- On top of all-data + the conditional fix, 90-day half-life trades headline bias for spread: + 3-month bias +0.19 (vs +0.12 unweighted) but the best 3-month overall score of any arm (0.239 + vs 0.290 unweighted, 0.359 benchmark), and pooled ALL −0.142. The 45-day dose is flat vs 90 + (ALL 4.038 vs 4.045) — the response is insensitive in this range. +- **Not defaulted** because the weights are shared-path: in prepost they nudge the placebo + headline bias *up* at all three campaign lengths (+0.09/+0.09/+0.02; score is a wash — 3-month + −0.14 better, 6-month +0.12 worse) — the wrong direction for the issue's own target metric. + Remains available where short-campaign spread matters more than bias purity. + +### Method notes +- The refactor no-op was verified against the benchmark before any A/B (all deltas 0.0), and the + uncentered→centred iteration was driven by the per-run correction logs — keep logging the + mean correction and the centred-out level; they made the wrong-sign diagnosis possible. +- Issue 13's "time-blocked baseline cross-validation" item shipped across F14/F15: the holdout + display is time-blocked (F14) and `_oof_baseline_predictions` gives every baseline row an + out-of-fold prediction (F15), shared by both calibration paths. + +--- + +## F14 — outcome-model fundamentals: `min_child_samples` 200→50 accepted; linear_tree, early stopping, calibration slope, seed ensembling and alternative learners rejected; the toggle headline bias localised to small-fit tree shrinkage (Issue 12) + +*2026-07-04 — Issue 12 verdicts. New module `benchmarking/baselines/power_model/fitting.py` +(time-blocked folds, calibration line, early-stopped capacity pick, and the model-factory registry +`OUTCOME_MODEL_FACTORIES`) behind four new `PowerModelMethod` knobs — `model_factory` (str or +callable), `n_seed_ensemble`, `early_stopping`, `calibrate_slope` — all defaulting to today's +behaviour (a default-config re-run diffs the benchmark at ≤0.001 pp everywhere). The baseline +holdout *diagnostic* switched from a shuffled 20% split to a time-blocked fold (Issue 13's honest +display item; estimates unchanged). Candidates A/B'd per the Issue 9 protocol: +`study_power_model_compare.py --method-overrides` placebo screens (`cp_0pct`, both modes), full +7-profile sweep for the survivor, all diffed against the post-F13 benchmark.* + +### Framing principles (recorded here so they aren't relitigated) +- **The objective targets the conditional mean.** The estimand is an energy ratio and energy is a + sum of conditional means, so L2 stays (design note §2). Power conditional on features is skewed + (near cut-in, around rated), so median-type objectives — MAE, Huber in its robust regime, + quantile-0.5 — estimate the median and bias the energy sum. The F5 shrinkage is a + *regularisation* artefact, not a loss artefact; changing objective does not fix it. Legitimate + within the mean family: Tweedie / variance-weighted L2 (efficiency candidates, untried). + Quantile objectives are out of scope for the point estimate (they return in Issue 19/WS4). +- **Tune on uplift metrics, never on prediction RMSE.** More regularisation can improve held-out + RMSE while worsening shrinkage. Yardsticks: placebo bias on the harness, per-bin residual + flatness, the predicted-vs-actual **calibration slope** (target ≈ 1) on a time-blocked held-out + baseline, and replicate spread. + +### ACCEPTED — `min_child_samples` 200→50 (now the driver default; benchmark regenerated) +- Placebo screen: prepost ALL Δscore **−0.62**, Δspread −0.75 (score cells 31 better / 9 worse); + toggle overall better at every campaign. Dose-response check at `min_child_samples=20` + overshoots (prepost ALL only −0.12, overall biases drift positive) — 50 is the sweet spot. +- Full 7-profile sweep: **prepost ALL Δscore −0.36, Δspread −0.41** (cells: score 216 better / 86 + worse of 469); overall P50 neutral-or-better at every campaign in both modes (prepost 3-month + placebo |bias| 0.121 → 0.018). The accepted cost: a few large toggle ti cells at 9/12 months + regress (9-month ti Δscore +3.2; toggle conditional mean +0.29 even though score cells split + 159 better / 145 worse) — accepted against the overall-P50/precision gains, the F13 precedent. +- Shipped as `TUNED_MODEL_PARAMS = {"min_child_samples": 50}` passed by the four HoT drivers; + the design-note common params in `make_outcome_model` (shared with the R-learner) are unchanged. + +### REJECTED — everything else, each with a one-screen placebo verdict +- **`linear_tree=True`** — mode-split (the F10 shear/veer pattern): toggle mildly better (ALL + Δscore −0.36) but prepost overall bias worse at every campaign (+0.32/+0.28/+0.11 pp) and + prepost conditional much worse (3-month ti Δscore +8.4). The hoped-for F5 edge-extrapolation fix + adds bias/variance under the prepost weather shift instead. +- **`early_stopping`** (time-blocked valid split, refit at the picked capacity) — neutral overall + in both modes *including the 3-month-toggle small-fit regime it was aimed at*; conditional + slightly net-worse; +~15% runtime. The fixed design-note capacity is validated across the ~10× + fit-size range. +- **`calibrate_slope`** — prepost neutral (nothing to correct), toggle **overcorrects**: bias + flips sign (3-month +0.39 → −0.38) with a spread cost (Δspread +0.60) — the F7 short-campaign + failure mode. Root cause: the line is fit on out-of-fold predictions from 80%-sized fits, which + shrink more than the 100% fit it is applied to, and that gap is largest exactly where the + correction is largest. +- **`n_seed_ensemble=4`** — overall deltas neutral in both modes (|Δspread| ≤ 0.02 pp) for ~75% + more runtime: replicate spread is dominated by weather sampling across windows, not seed noise. +- **`model_factory="hgb"`** (sklearn HistGradientBoostingRegressor, capacity-matched) — prepost + worse (ALL Δscore +0.26), toggle marginally better: LightGBM's behaviour is family-level, not an + implementation quirk. Stays in the registry as the cross-implementation check. +- **`model_factory="linear"`** (impute→scale→Ridge structured baseline) — far worse overall as + expected (prepost placebo bias +3.5/+2.5/+1.3 pp; misspecification dominates the hoped-for + "shrinkage-free" property), **but the cross-check paid off** (next section). + +### The toggle headline bias is small-fit tree shrinkage — three independent probes agree (Issue 13 hand-off) +- **Calibration slopes scale with fit size**: ~1.0005–1.005 at n≈100k (2-year prepost baseline), + ~1.008–1.017 at n≈12–25k, **~1.013–1.027 at n≈6k** (3-month toggle off rows). The prepost + baseline is essentially slope-calibrated, so the −0.4 pp prepost headline bias is *not* a global + calibration artefact — consistent with Issue 13's covariate-shift mechanism for prepost. +- **The linear model's toggle headline reads ~0** (3-month +0.39 → −0.01, better at every + campaign length): a learner with no tree-style shrinkage does not show the bias. +- **More training data removes it**: `toggle_campaign_only=False` (the Issue 13 2×2's + {all-data, no-calibration} cell, measured) cuts the 3-month toggle headline bias to +0.07 — + but without the Issue 13 calibration/time-feature guards it imports drift everywhere else + (conditional 3-month ti Δscore +12.7; ≥6-month campaigns uniformly worse; cells 13 better / 58 + worse). Campaign-only stays the default; revisit inside Issue 13's full 2×2. +- Capacity is *not* the lever: the accepted `min_child_samples=50` barely moves the 3-month + toggle headline (+0.39 → +0.35). The shrinkage that matters is intrinsic to boosted trees on + ~6k rows, so Issue 13's data-side fixes (more rows, made safe by calibration) are the right + attack. + +### Method notes +- The placebo screen again did all the discriminating; nothing needed the full sweep to be + rejected. One screen ≈ 13 min vs ≈ 55 min for a full sweep. +- Post-hoc OOF calibration applied to a refit-on-100% model is structurally biased toward + overcorrection at small n (the OOF-vs-final shrinkage gap) — any future residual-calibration + design (Issue 13) must estimate the correction against the *final* model's predictions, e.g. + by calibrating on a window the final model genuinely never saw, not by recycling OOF folds. + +--- + +## F13 — removal ablation: dropping the availability feature + five ERA5 columns improves the benchmark; low importance ≠ removable + +*2026-07-04 — follow-up to the Issue 9–11 additions: the same A/B protocol run in reverse (remove each +currently-accepted feature group, keep any removal that noticeably improves the score). Two new +ablation knobs on `PowerModelMethod`: `era5_exclude` (drops raw ERA5 columns + their sin/cos +companions; guarded against excluding `matching_vars` while `conditional_uplift` is on) and +`availability_feature` (drops the per-reference availability *feature*; `availability_col` stays +required for the downtime filter). Screens on the placebo, confirmation via two full sweeps +(`--method-overrides` on `study_power_model_compare.py`), all diffed against the post-F12 benchmark.* + +### The accepted removal set (now the driver default; benchmark regenerated) +`availability_feature=False` + `era5_exclude = CURATED_ERA5_EXCLUDE = (apparent_temperature, +dew_point_2m, precipitation, rain, snowfall)`. Full-sweep deltas (pp; negative = better): +- **prepost overall**: Δ|bias| −0.19, Δspread −0.18, Δscore **−0.24** — the largest overall + improvement of the whole Issue 9–11 campaign, and it came from *removing* features. Per campaign + (placebo): 3 mo bias −0.59 → −0.12, 12 mo −0.21 → **0.00**, 6 mo overshoots mildly (−0.17 → +0.30). +- **toggle**: overall neutral (≤0.006); conditional Δ|bias| −0.98, Δscore **−1.47**. +- **prepost conditional**: the one cost, Δscore +0.54 (cells split 212 better / 183 worse) — accepted + against the overall-P50 gains (Phase 1 is judged on P50 accuracy/precision first). +- Availability-alone (set A) shows nearly the same numbers; the ERA5 trims add a small consistent + extra (prepost conditional Δ|bias| +0.07 → −0.17 vs A). The removals stack cleanly — unlike the + F12 max+min interaction. + +### Why removing availability helps +References are almost always available, and when one is not, its *power* column already carries the +fact (0/NaN) — so the counter added noise and a mild maintenance-calendar proxy rather than wake +information. The curated-feature physical argument ("the model should know whether a reference is +waking") was measured and lost to the data. + +### Kept columns — low importance is not a removal licence +Removing bottom-of-the-ranking columns often *hurt*: `pressure_msl` removal cost +2.09 pp prepost +conditional score (the model evidently uses the msl-vs-surface pressure pair jointly), `weather_code` +removal +0.49, `cloud_cover` removal +0.39 toggle conditional, humidity removal worse everywhere. +Together with F12's rank-3/rank-4 accepted/rejected split, the lesson is symmetric: importance rank +predicts neither a feature's value nor its removability — only the benchmark gates do. + +--- + +## F12 — reference active-power **minimum** accepted as a default feature; SD and max rejected (Issue 11) + +*2026-07-04 — Issue 11 verdicts. Candidates A/B'd one field at a time per the Issue 9 protocol: +`study_power_model_compare.py --method-overrides '{"reference_stat_cols": [...]}'` against the committed +benchmark — placebo (`cp_0pct`, both modes) screen first, full 7-profile sweep for survivors. The HoT +loader now also unpacks `wtc_ActPower_max` / `wtc_ActPower_min` (the SD was already loaded); the fields +reach the model via `build_reference_features(..., extra_cols=...)` / +`PowerModelMethod.reference_stat_cols`.* + +### Verdicts (deltas vs the pre-change benchmark, in pp; negative = better) +- **`wtc_ActPower_min` — ACCEPTED, now the default** (`reference_stat_cols=("wtc_ActPower_min",)` in the + HoT drivers). Better on every gate in both modes: overall P50 Δ|bias| −0.01 (prepost) / −0.006 (toggle), + Δspread ~0 / −0.009; conditional mean Δ|bias| **−0.46 (prepost)** / **−0.75 (toggle)**, Δscore −0.40 / + −1.30. Interpretation: the within-period minimum tells the model when a reference dipped (gust lulls, + brief curtailments) — a farm-sited variability signal the mean hides. **Benchmark JSON regenerated** + from this configuration's full sweep. +- **`wtc_ActPower_stddev` — REJECTED.** The placebo screen alone disqualified it: conditional mean + Δ|bias| +2.8 / Δscore +3.98 (prepost), +0.65 / +1.38 (toggle), and it jumped to importance rank 5. + The within-period power SD is exactly the kind of channel the placebo gate exists for. +- **`wtc_ActPower_max` — REJECTED.** Full sweep: it shifts the prepost overall bias **uniformly +0.33 pp + across all seven profiles** — a counterfactual level shift, not uplift tracking. That happens to offset + the structural −0.4 pp prepost headline bias (|bias| improves) but costs spread (+0.08; 240/469 cells + worse) and is an accidental cancellation — Issue 13 addresses that bias properly. Toggle-side it helps + (−1.13 conditional), but combined with `min` (trio minus SD) it is toxic: prepost conditional + Δ|bias| +3.4 / Δscore +5.2. Max and min together destabilise the matched conditional fits. +- **Reference nacelle wind speed / wind-speed SD — rejected without trial** (recorded per the issue): a + reference anemometer is at high risk of calibration drift, which a prepost campaign reads as uplift; + reference *power* is the calibration-stable channel, and same-type references degrade like the test + turbine, giving a fairer counterfactual expectation. + +### Method note +Importance rank alone was a poor red-flag here: `max`/`min` both ranked 3rd–4th (gain frac 3–4%), yet one +was accepted and one rejected — the benchmark gates (placebo bias, spread, conditional cells) did the +discriminating, not the ranking. + +--- + +## F11 — explicit time features rejected: campaign-drift, season and solar all fail or add nothing (Issue 10) + +*2026-07-04 — Issue 10 verdicts. `benchmarking/baselines/time_features.py` ships the features +(`days_since_campaign_start`, June-21-anchored `season` sin/cos, NOAA `solar` altitude/azimuth validated +against an ephem reference to <0.01°) behind `PowerModelMethod.time_features` (+ `latitude`/`longitude`), +default **off**. A/B'd one at a time on the placebo (`cp_0pct`, both modes) via +`study_power_model_compare.py --method-overrides`.* + +### Verdicts (placebo deltas vs benchmark, pp) +- **`days_since_campaign_start` — REJECTED.** Prepost is the anticipated failure, measured: overall + Δbias +0.23, Δspread **+1.08**, Δscore +0.97; conditional Δscore +5.3 with 54/67 cells worse. Trees + cannot extrapolate the feature past the changeover (every upgraded-row value exceeds the training + range, so predictions clamp at boundary leaves), which *adds* variance instead of absorbing drift. + Toggle is ~neutral (−0.001 overall) — but prepost and toggle share one code path, so it stays out. +- **`season` — REJECTED.** Prepost overall slightly worse (Δscore +0.14, max Δ|bias| +0.26 at one + campaign length), conditional worse in both modes (+0.39 / +0.16). Against a <12-month baseline the + pair is a partial calendar proxy, as the issue warned. +- **`solar` — REJECTED.** Neutral overall (≤0.03) but conditional worse in both modes (+0.90 / +0.29 + Δscore) and negligible importance (rank 22–24 of 32, gain frac ≤0.0003) — the instantaneous weather + columns already carry the diurnal signal at this site. + +### Method note +The feared "time feature dominating the importance ranking" red flag never fired — all three sat far +down the ranking (gain frac ≤0.0006) *while still doing damage through spread*. The placebo benchmark +gate, not the importance watch, is the effective detector for drift-importing features. The module stays +in the tree for future sources (e.g. a site with genuine reference drift may re-litigate +`days_since_campaign_start` in toggle-only form). + +--- + +## F10 — ERA5 derived quantities: utility shipped; hub-height wind speed validated standalone; no derivation earns a default place in the full model (Issue 9) + +*2026-07-04 — Issue 9 verdicts. `benchmarking/baselines/era5_derived.py` is the shared derivation +utility (shear exponent, hub-height ws via the shear power law + `hub_height_m` — `HOT_HUB_HEIGHT_M = +59.0`, gust ratio, gust margin, veer, moist-air density), reused by +`inspect_era5_matching_importance.py` and available to the CEM matching step; features reach the model +via `PowerModelMethod.era5_derivations`, default **off**. Screened per candidate on the placebo, full +sweeps for survivors (`study_power_model_compare.py --method-overrides`).* + +### The gust "TI proxy" is not one (measured against real SCADA TI) +- `gust_ratio = wind_gusts_10m / wind_speed_10m` correlates with the test turbine's measured TI at + **Pearson +0.03** (Spearman +0.11; T01, ws>4 m/s, n≈195k) and implies TI ≈ 0.31 at the median vs the + real 0.17 — ERA5's hourly grid-scale gustiness is a different quantity from local 10-min turbulence. +- Variants (per the issue's "play around" instruction): the absolute **gust margin** `gusts − ws_100m` + is the best simple correlate (+0.22), shear exponent −0.21, |veer| −0.21, `gusts/ws_100m` +0.11. + A LightGBM fit of TI on *all* ERA5 columns reaches held-out **R² = 0.38**, dominated by wind-direction + sin/cos (~31% of gain — wake/terrain sectors) — much of site TI is direction-determined and the model + already sees direction. + +### A/B verdicts (vs benchmark, pp) +- **`wind_speed_hub` — validated, left opt-in.** With reference features removed (ERA5-only ranking, the + Issue 9 exploration) it *dominates*: 63% of gain, permutation importance 0.48 vs 0.10 for raw + `wind_speed_100m`. In the full model its full sweep improves overall P50 in both modes (prepost + Δ|bias| −0.02, Δspread −0.035 uniformly across profiles; toggle conditional −1.02 score) at a small + prepost conditional cost (+0.13). But combined with the accepted `wtc_ActPower_min` its marginal value + disappears (combo no better than `min` alone, toggle conditional diluted), so it is **not defaulted**; + it is the natural candidate for the F6 `matching_vars` revisit and for reference-poor sources. +- **`shear_exponent`, `veer` — REJECTED (mode-split).** Both help toggle conditional (−0.41 / −0.47 + score) and hurt prepost (+0.80 / +0.54); one code path, so out. +- **`gust_ratio`, `gust_margin`, `air_density` — REJECTED.** Overall neutral; conditional worse + (gust_ratio prepost +1.33 score with the (6, ws) cell +5.9; gust_margin +0.34/+0.23; air_density + +0.60 prepost, 25/35 cells worse). Consistent with the TI-proxy result: these columns add split noise, + not cause. + +### Interpretation +The reference active-power features already carry the site signal ERA5 derivations try to reconstruct — +in the full model every derivation lands at gain frac ≤0.0012. ERA5 derivations matter where references +are absent: the ERA5-only fit (R² 0.85) is where `wind_speed_hub` shines, which is exactly the CEM +matching / AEP-extrapolation context (Issues 8/15), not the counterfactual feature set. + +--- + +## F9 — the matched two-direction conditional cross-prediction shipped as the sole conditional method, on by default + +*2026-07-03 — Issue 8 ship. The F7/F8 development-time A/B flag `bias_correct` is **removed**; the +matched two-direction cross-prediction is now the sole conditional-uplift path, controlled by +`PowerModelMethod.conditional_uplift: bool = True` (**default on**). Current helpers: +`PowerModelMethod._estimate_conditional` (ERA5 match + forward/reverse fits) and `_conditional_by_bin` +(the re-leveled per-bin shape); the pure re-level helper `_relevel_conditional` is unchanged. Re-run +via `study_power_model_compare.py` (Issue 7), which overlays the committed benchmark vs the current run +vs truth per covered `(profile, condition)`.* + +### What shipped (supersedes the F8 "still opt-in" decision) +- **Default flip.** F8 left the correction opt-in pending an A/B verdict; that verdict came in and it + became the default. `conditional_uplift=False` still skips the expensive cross-prediction and returns + overall-P50 only, so the opt-out remains for the ERA5-less / overall-only configuration. +- **Overall P50 unchanged, by construction.** The headline is still the single full-window fit + (F8): the ship is bit-identical on overall P50 (max |Δ bias| ≤ 1e-4 pp both modes) — the correction + is spent only on the per-condition decomposition. +- **Conditional accuracy roughly halved.** Over the condition-dependent + placebo profiles, mean per-bin + |bias| falls **prepost 18.2 → 6.3 pp**, **toggle 13.0 → 4.1 pp** (score prepost 22.6 → 10.2, toggle + 18.2 → 8.0); ~87% of covered bins improve. The remaining worse bins are the rare sparse tails (tiny + counts, both methods noise) — the F7 sparse-extreme overshoot, left unfloored by choice (F8). + +### Packaging +- **One run folder**, not four: conditional CSVs under a `conditional/` subfolder, the implied-shrinkage + diagnostic under `plots/7_conditional_uplift/`. The `implied_shrinkage` diagnostic stays on the public + surface; the "bias correct(ed)" naming is gone. +- **Benchmark JSON regenerated** under the new default; `docs/v1/issues.md` Issue 8 updated to match. + +### Implications / follow-ups (carried from F7/F8) +- A per-reporting-bin matched-count floor for the sparse-extreme overshoot (the handful of worse bins). +- Density-ratio weighting (WS4) to enlarge the short-campaign matched set. + +--- + +## F8 — the self-consistent correction: keep the uncorrected full-data headline, re-level the matched decomposition onto it — overall now no worse + +*2026-07-02 — Issue 8, the fix for F7's headline cost. Same A/B command +(`study_power_model_compare --modes prepost --profiles cp_0pct ti_dependent_cp ws_dependent_cp +--bias-correct`) and artefacts as F7. Estimator reworked in `PowerModelMethod._estimate_bias_corrected` +/ `_corrected_conditional`; new pure helper `_relevel_conditional`; `energy_ratio_by_bin` now also +returns per-bin `sum_actual` / `sum_counterfactual`.* + +### The estimator (final) +- **Overall = the uncorrected full-data estimate.** Train on **all** baseline, predict **all** upgraded, + one energy ratio — *identical* to the `bias_correct=False` headline (a unit test asserts exact + equality). The whole-window shrinkage integrates to ≈ 0 (F5), so this is already the cleanest overall; + the correction is spent only on the decomposition. +- **Per-bin = the matched two-direction shape, re-leveled onto the headline.** The shape + `1+u_b = sqrt((1+r_fwd_b)/(1+r_rev_b))` still comes from the **CEM-matched** forward/reverse fits + (matching is required — without it the reverse model predicts out-of-distribution across the prepost + weather shift). Each condition is then rescaled by one factor `λ_c = (1+overall)/(1+u_agg)` whose + **weights are the full-upgraded per-bin energy**, so the reported per-bin MWh partitions the full-data + headline exactly ("overall = aggregation" self-consistency). + +### Why not the two earlier variants (both measured) +Getting here took ruling out two tempting overalls, both worse than uncorrected: +- **Global `sqrt`-combine** (the F7 estimator): mean overall |bias| **0.49 pp** — not self-consistent + (three separate nonlinear reductions) and still worse than uncorrected at short campaigns. +- **Matched forward-only** `r_fwd_all`: mean overall |bias| **0.82 pp** — *worse still*. The premise + "the matched forward recovers the overall because shrinkage integrates out" was wrong: on the ≈ 11% + matched subset the forward model is not energy-conserving, so `r_fwd_all` keeps a residual `(1+u)/s` + shrinkage (~+1 pp at 3 mo). The `sqrt`-combine had actually been *cancelling* that. Lesson: the + matched subset can't beat the full-data overall for the headline; only the per-bin *shape* needs the + matched cross-prediction. + +### Result +- **Overall: no worse — identical to uncorrected** at every profile × campaign (|Δ bias| and |Δ spread| + ≤ 1e-4 pp, float noise). Contrast: uncorrected 0.35 pp, F7 combine 0.49, forward-only 0.82. +- **Per-bin de-tilt preserved:** 57 better / 9 worse / 3 ~ of 69 covered cells; mean per-bin |bias| + ≈ 14 pp → ≈ 8.5 pp — the F5 tilt is gone across the populated range, and the ws & ti decompositions + now energy-aggregate back to the (unchanged) headline. +- The 9 "worse" cells remain the sparsest TI/ws extremes (overshoot), left **unfloored** by choice; the + re-level keeps them from moving the headline, so it is a tail display issue, not a headline bug. + +### Decision / implications +- **`bias_correct` becomes a pure decomposition refinement on an unchanged headline.** The correction + never touches the P50 — it only re-attributes the same total MWh across bins with the shrinkage tilt + removed. That is a safe, Pareto-neutral property for the overall. +- **Still opt-in (`bias_correct=False` default);** whether to make it the default is a later call. +- **Follow-ups:** a per-reporting-bin matched-count floor for the sparse-extreme overshoot (one-line); + density-ratio weighting (WS4) to enlarge the short-campaign matched set; a toggle-mode A/B (prepost is + the F5/F7/F8 case reported here). + +--- + +## F7 — the two-direction bias correction removes the F5 per-bin tilt but costs overall P50 at short campaigns, so it stays opt-in + +*2026-07-02 — Issue 8, Component 6 A/B run. Command: +`study_power_model_compare --modes prepost --profiles cp_0pct ti_dependent_cp ws_dependent_cp --bias-correct`. +The committed benchmark is the **uncorrected** frozen run, so the run's before/after machinery reads as +corrected ("current") vs uncorrected ("benchmark") vs truth. Artefacts under the run's `comparison/`: +`conditional_before_after__.png` (per-bin overlays), +`conditional_benchmark_comparison_prepost.csv` (per-bin |bias| verdict) and `benchmark_comparison_prepost.csv` +(overall). Single-case overlays also reproduced via `inspect_prepost_hard_case.py` (now runs uncorrected + +bias-corrected side by side).* + +### Observation — the per-bin conditional bias is largely removed (the F5 target) +Across the two condition-dependent hard cases plus the placebo, at the 12-month campaign the correction +flattens the per-bin uplift toward truth in the large majority of bins: **57 better / 9 worse / 3 neutral of +69 covered cells**, mean per-bin |bias| ≈ **14 pp → ≈ 8.5 pp**. The F5 tilt is gone across the populated +range — e.g. `cp_0pct` placebo `ti (0.30,0.35]` 17.6 → 0.04 pp and `ws (2,4]` 45.6 → 12.7 pp; +`ws_dependent_cp` `ti (0.30,0.35]` 18.7 → 0.17 pp. The implied shrinkage the correction cancels is ≈ 0.99 +overall on these cases (a small overall compression that integrates to ≈ 0, exactly as F5 predicted). + +### Observation — the cost: overall P50 degrades at short campaigns +The correction is **not** free on the headline number. Overall |bias| / spread (pp), corrected vs uncorrected: + +| campaign | corrected \|bias\| | uncorrected \|bias\| | Δ\|bias\| | Δ spread | +| --- | --- | --- | --- | --- | +| 3 mo | 0.66–0.78 | 0.18–0.63 | **+0.06 … +0.16** (worse) | ≈ +0.6 (worse) | +| 6 mo | 0.56–0.73 | 0.18–0.19 | **+0.38 … +0.54** (worse) | ≈ +0.05 (worse) | +| 12 mo | 0.05–0.10 | 0.23–0.24 | **−0.14 … −0.18** (better) | ≈ −0.15 (better) | + +So the done-when "overall P50 no worse" holds **only at 12 months**; at 3/6 months the correction adds a +few tenths of a pp of bias and spread. + +**Mechanism — finite-sample cost of a nonlinear two-ratio combine on a shrunken matched set.** The corrected +overall is a *different estimator*, not the same one on less data: the uncorrected path is one energy ratio +`Σactual/Σcf − 1` over **all** upgraded rows from **one** model trained on **all** baseline rows; the corrected +path is `sqrt((1+r_fwd)/(1+r_rev)) − 1` from **two** models trained on the **CEM-matched subset**. Two things +compound at short campaigns: (1) the matched subset collapses — the short upgraded window can only match a +sliver of the abundant baseline, so from the per-run CEM balance the fraction of baseline actually used falls +to **≈ 11% at 3 mo** (~11k rows/side) vs **≈ 47% at 12 mo** (~47k), i.e. each counterfactual model trains on +~9× less data at 3 mo; (2) the two noisier per-direction ratios are combined through a **nonlinear** function, +so by Jensen's inequality the expected combined value is offset from the noise-free value, and that offset +**grows as the inputs get noisier**. The campaign-length signature is the tell: a change of *estimand* (matching +to a different weather mix) would be roughly campaign-independent, but this cost **vanishes as data grows** +(worse at 3 mo → better than uncorrected at 12 mo) — the fingerprint of a finite-sample effect. The uncorrected +overall pays none of this and is already clean because the shrinkage integrates to ≈ 0 over the whole window +(F5): **the correction spends precision fixing a per-bin problem the headline never had.** + +### The failure mode — sparse extreme bins overcorrect +All 9 "worse" per-bin cells are the sparsest condition extremes (lowest-TI `(0.0,0.05]`, highest-TI +`(0.40,0.50]`, lowest-ws `(2,4]`). There the two-direction ratio is estimated on very few matched rows, so +the correction overshoots — e.g. `ti (0.45,0.50]` swings −76.7 → +92.9 pp. This both drags the mean per-bin +|bias| up (hence ≈ 8.5 pp, not lower) and, via the extreme bins, perturbs the overall energy ratio at short +campaigns. Some extreme bins also drop to `NaN` (the non-positive-ratio / thin-bin guard), visible as gaps +in the single-case overlays. + +### Decision / implications +- **Stays opt-in (`bias_correct=False` default); no default flip.** It decisively fixes the per-bin + decomposition but is not a strict Pareto improvement on the overall P50, so it is not ready to be the + default — exactly the A/B outcome the opt-in design was built to allow. +- **Candidate follow-ups** (being taken up next): the defect has two distinct sources needing two distinct + levers, since the overall is a **global** energy ratio (sparse reporting bins carry little weight in it, so a + per-bin fix does **not** touch it): + 1. **Per-bin extreme overshoot** → a **minimum matched-count floor per reporting bin**, below which that bin + falls back to the uncorrected estimate — directly kills the sparse-bin overshoot (the 9 "worse" cells). + 2. **Overall short-campaign cost** → report the **uncorrected single-direction estimate as the headline P50** + (F5: the whole-window shrinkage integrates to ≈ 0, so the two-direction combine only adds finite-sample + noise there), keeping the two-direction correction for the per-bin *decomposition* where the shrinkage does + **not** integrate out. This makes "overall no worse" hold by construction. + Density-ratio weighting instead of hard CEM subsampling (earmarked for WS4) would additionally reclaim the + short-campaign matched-sample size. +- Toggle mode not yet A/B'd; prepost is where F5 was diagnosed and is the case reported here. + +--- + +## F6 — the ERA5 matching variables for Issue 8 bias-cancellation are `wind_speed_100m` + `wind_gusts_10m` + `wind_direction_100m` + +*2026-07-02 — Issue 8, Component 1 matching-variable analysis. New one-off script +`benchmarking/baselines/inspect_era5_matching_importance.py` ranks the ERA5-only fields by how well +they predict the test turbine's real (un-upgraded) power, using LightGBM gain and held-out sklearn +permutation importance. Run on `T01`, ~253k normally-operating rows over the default 2016–2020 HoT +window; outputs (`feature_importance.png`, `predicted_vs_actual.png`, `era5_matching_importance.csv`) +under `/inspection_era5_matching`.* + +### Observation +The ERA5→test-power model predicts well (held-out **R²=0.84**, RMSE ≈ 287 kW), so the ranking is +trustworthy, not just precise. Wind-magnitude fields dominate: `wind_speed_100m`, `wind_speed_10m` +and `wind_gusts_10m` together carry ≈ 89% of the gain; every direction / thermodynamic field is +< 2% on both views. But `wind_speed_10m` and `wind_speed_100m` are strongly collinear, so **gain and +permutation disagreed** on the 2nd variable (gain favoured `wind_speed_10m`, permutation favoured +`wind_gusts_10m`) — the collinearity makes neither view a clean guide. + +### What was done +Folded the two collinear speeds into one physical vertical-shear exponent +`alpha = ln(ws_100m / ws_10m) / ln(100/10)` (a stability / turbulence proxy that directly attacks the +F5 cause) and dropped `wind_speed_10m`. Held-out fit was **unchanged** (R²=0.844), confirming the two +speeds were substitutes, and with the redundancy gone **gain and permutation agree cleanly**: +`wind_speed_100m` ≫ `wind_gusts_10m` ≫ `wind_shear_exponent`, then a flat tail. The shear exponent is +a real independent signal (~2.5% on both views) but an order of magnitude below the two magnitude +fields at HoT. + +### Decision +- **Matching set = `("wind_speed_100m", "wind_gusts_10m", "wind_direction_100m")`.** The first two are + the top of the importance ranking (once the 10m/100m redundancy is folded out); **wind direction is + added on physical grounds** — it governs the wake geometry between the test turbine and its + references, so omitting it would leave a first-order confounder unmatched even though its *marginal* + importance for predicting power is small. +- **Shear exponent deferred.** It ranks 3rd on importance and modest, and adopting it properly means + adding the derivation to the method's real feature path (`power_model/features.py`), not just the + analysis script where it currently lives. Left for another day; the shear derivation stays in + `inspect_era5_matching_importance.py` only and changes no scored method or benchmark. + +### Bin widths — verified on real HoT by the coverage sweep +A CEM coverage/sensitivity sweep (T01 prepost split at 2018-06-01, ~121k baseline / ~132k upgraded +normally-operating rows, via `benchmarking.baselines.power_model.matching.coarsened_exact_match`) +confirmed 3-var matching is affordable despite the nominal cell explosion — HoT weather concentrates +(prevailing SW ≈ 240°), so occupied cells are a small fraction of nominal and retention stays high: + +| ws / gust / dir | one-sided dropped | matched/side | retained base / up | +| --- | --- | --- | --- | +| 2 / 3 / 30° | 101 | 111,166 | 91.7% / 84.4% | +| **2 / 3 / 20°** | **143** | **109,313** | **90.1% / 83.0%** | +| 2 / 3 / 10° | 271 | 105,511 | 87.0% / 80.1% | +| 1 / 3 / 5° | 950 | 95,930 | 79.1% / 72.9% | + +**Chosen widths: `wind_speed_100m` = 2 m/s, `wind_gusts_10m` = 3 m/s, `wind_direction_100m` = 20°.** +ws = 2 m/s (generalises fine over that band and retains a touch more than 1 m/s); direction = 20° +because `wind_direction_100m` is a *reanalysis* direction — spatially smooth and coarse, so finer than +~20–30° is finer than the signal supports, and 5–10° roughly 3–7×'s the dropped one-sided cells for +little directional gain. 20° keeps ~90%/83% retention with ~120 matched rows per two-sided cell. + +### Implications +Component 3 hard-codes the method's `matching_vars` default to the 3-var set and `matching_bin_edges` +to `{wind_speed_100m: 0..32 @ 2, wind_gusts_10m: 0..44 @ 3, wind_direction_100m: 0..360 @ 20}` (fixed +sectors, no wraparound — adjacent sectors are just separate cells, per Component 2). Per-farm re-tuning +and a proper shear feature are later steps. + +--- + +## F5 — power_model's condition-dependent uplift error is the counterfactual model's own conditional bias (shrinkage), not a §3 post-treatment-conditioning artefact + +*2026-07-01 — first result off the conditional-uplift instrument (per-(ws, TI)-bin scoring; see +`benchmarking/harness/conditions.py`, `scoring.py`, `PowerModelMethod._conditional_uplift`). +Diagnosed with a new per-segment residual diagnostic +`benchmarking/baselines/power_model/diagnostics.py::_plot_residual_binned` (writes +`residual_binned.png` and `residual_binned_pct.png` under `5_uplift_modelling/`), regenerated via +`benchmarking/baselines/inspect_prepost_hard_case.py`. The pinned case is the `cp_0pct` placebo on +`T07`, 6-month prepost, true uplift 0% — so in both the held-out baseline and the "upgraded" window +the residual is pure model error, an unusually clean read on model bias.* + +### Observation +`power_model`'s overall P50 is excellent (F3/F4), but its **per-bin** uplift decomposition is badly +distorted at the condition extremes. On `ti_dependent_cp` the recovered uplift slopes to ≈ −75 pp in +the highest TI bin where the truth is roughly flat; on `ws_dependent_cp` the (2,4] m/s bin reads +≈ −30 pp against a +17 pp truth. The overall estimate is unaffected because these errors integrate +to ≈ 0. + +### Evidence — three facts that localise the cause +1. **Not a §3 / binning-axis problem.** The synthetic upgrades for `ti_dependent_cp` and + `ws_dependent_cp` use `ws_delta = 0` (`ws_factor = 1.0`) and never modify `wind_speed_sd` + (`generator.py` `modified_columns = active_power, gen_rpm, wind_speed`). So the test turbine's + **measured** ws/TI under treatment equals its **original untreated** ws/TI exactly — the method + already bins on the same treatment-invariant axis the ground truth uses. (Binning the estimate on + a reference-derived ws/TI instead would therefore change nothing; confirmed by reasoning, not + pursued. Ground-truth binning was left unchanged.) +2. **Not confounding.** Toggle (concurrent reference, ≈ no temporal confounding) shows the *same* + per-bin distortion shape as prepost. +3. **It is model shrinkage / conditional bias.** On the placebo, `residual_binned.png` shows the mean + residual (actual − predicted) tilting from ≈ −25 kW at low power to ≈ +80 kW at high power — the + tilt survives the Bland-Altman `mean(actual, predicted)` axis, so it is a real conditional bias, + not just the errors-in-variables inflation from binning by actual. Across TI the same residual runs + ≈ +30 kW → −38 kW (zero-crossing at TI ≈ 0.17), exactly the shape of the fake TI-uplift. The + baseline and upgraded residual curves overlay (as they must at truth 0). + +### Root cause +A regularised learner minimises squared error, so its prediction is pulled toward the conditional +mean: with imperfect features it **predicts smoother than reality** — over-predicts where power is low, +under-predicts where it is high (predicted-vs-actual slope < 1). The §3 rule forbids the test +turbine's own ws/TI as features, so the counterfactual model carries no direct turbulence +information; its residuals are therefore correlated with TI. Because power maps monotonically to wind +speed, and TI is inversely related to wind speed at fixed power, this single compression re-appears as +a negative residual at low ws / high TI and positive at high ws / low TI. The compression averages to +≈ 0 over the whole window (so the headline uplift is clean), but slicing the energy ratio +`Σactual / Σcounterfactual` **by condition** re-exposes the conditional bias as a spurious +condition-dependent uplift. + +A **second, separate** failure mode inflates the visible excursions at the extremes: the low-ws / +high-TI bins hold very little energy, so a small kW residual over a tiny `Σcounterfactual` becomes a +huge *percentage*. The `residual_binned_pct.png` view (each bin's residual as a % of that bin's mean +power) makes this explicit — the well-populated bins sit within ±5–15% while the tiny-power bins blow +out to −160% (ws 5 m/s) / −330% (TI 0.35). So the extremes combine real conditional bias with ratio +instability. + +### Interpretation +The per-bin instrument is faithfully measuring **model bias**, not a defect in the harness or the +conditioning axis. The overall-P50 verdict of F3/F4 stands; what F5 adds is that *conditional* P50 is +only trustworthy where (a) the counterfactual model's conditional bias is small and (b) the bin holds +enough energy for a stable ratio. + +### Implications / candidate directions (not yet actioned) +1. **Reduce the counterfactual model's conditional bias** — the core lever. Leading candidate: + **baseline-residual calibration** — estimate the model's per-condition mean residual `b(cond)` on + untreated data (prepost baseline / toggle off-blocks, where the true uplift is 0) and subtract it + from the upgraded per-bin ratio. On the placebo the baseline residual curve *is* an estimate of + that bias and overlays the upgraded curve, so this should flatten the conditional-uplift curves + toward truth. To be designed before coding. +2. **Give the model a treatment-invariant turbulence proxy** (each reference's own sd/ws, or ERA5 + gust/spread) so the counterfactual can learn the TI–power relation without touching the treated + signal (§3-legal). Partial: only as good as the proxy's correlation with local TI. +3. **Guard the fragile tails** — suppress or flag bins below an energy/count floor; orthogonal to + 1–2 and fixes only the ratio-instability half. +4. **Reporting/tooling shipped this cycle:** the two `residual_binned*` diagnostics (shared y per row; + percentage version normalised by each bin's own mean power, with the shared y-axis sized from bins + within ±30% so tiny-power outliers clip rather than crush the scale). Widening TI bins 5%→10% was + tried to tame the plot and **reverted** — it did not address the underlying bias and the y-axis + sizing is the better fix. + +--- + +## F4 — power_model beats v0 in prepost and at longer-toggle; v0's edge is only short-toggle, and v0 alone breaks on rated-power uprates + +*2026-06-30 — the three-way comparison F3 flagged as the open question (power_model vs v0 in +*both* modes). Source: `benchmarking/baselines/study_power_model_compare.py`, which re-runs +**only** `power_model` over the current overnight cases and merges it with the frozen `v0_binned` ++ `naive_ratio` rows from the overnight run (`study_overnight_{prepost,toggle}.py`, +`include_v0=True`). Seven `overnight_profiles` (`cp_minus_10pct`, `cp_0pct`, `cp_plus_3pct`, +`cp_plus_10pct`, `ws_dependent_cp`, `ti_dependent_cp`, `rated_plus_5pct`), `n_replicates=4`, +seed 0; prepost campaigns 3/6/12 mo (84 cases/method), toggle 3/6/9/12 mo (112 cases/method). The +script's alignment guard confirmed all 84 + 112 fresh cases match the reference run's +method-independent ground truth exactly, so the merge compares identical cases.* + +All numbers below are percentage points of fractional uplift (a 0.01 fraction = 1 pp). Bias = mean +signed error, spread = std of signed error, RMSE pooled over all cases for that mode/method. + +### Observation — pooled over all profiles and campaign lengths + +| mode | method | bias | spread | MAE | RMSE | within ±1pp | per-case win | +|---|---|---|---|---|---|---|---| +| prepost | naive_ratio | −1.11 | 4.62 | 3.83 | 4.73 | 17% | 0% | +| prepost | **power_model** | **−0.39** | **0.49** | **0.53** | **0.62** | **86%** | **73%** | +| prepost | v0_binned | −0.73 | 0.72 | 0.84 | 1.03 | 62% | 27% | +| toggle | naive_ratio | +0.15 | 0.16 | 0.19 | 0.22 | 100% | 39% | +| toggle | power_model | +0.16 | 0.19 | 0.20 | 0.24 | 100% | 22% | +| toggle | v0_binned | −0.01 | 0.33 | 0.23 | 0.32 | 98% | 38% | + +("per-case win" = share of the N cases where that method has the smallest |error|.) In prepost, +**power_model's |error| is smaller than v0's in 73% of cases and smaller than naive's in 100%**. + +### Observation — RMSE by campaign length (the short-data story, serves G2) + +| mode | method | 3mo | 6mo | 9mo | 12mo | +|---|---|---|---|---|---| +| prepost | naive_ratio | 7.32 | 3.19 | — | 1.84 | +| prepost | **power_model** | **0.82** | **0.46** | — | **0.53** | +| prepost | v0_binned | 1.22 | 0.94 | — | 0.89 | +| toggle | naive_ratio | **0.30** | 0.23 | 0.17 | 0.14 | +| toggle | power_model | 0.38 | **0.22** | **0.18** | **0.11** | +| toggle | v0_binned | 0.32 | 0.31 | 0.33 | 0.32 | + +The two structural facts: **(a)** in prepost power_model leads at every length, its biggest margin +at 3 months (0.82 vs v0 1.22 vs naive 7.32); **(b)** in toggle, power_model and naive both tighten +with more data (power_model 0.38 → 0.11), but **v0 does not improve with campaign length** — it +sits at ~0.32 RMSE from 3 to 12 months. So v0 only wins the shortest toggle campaign; from 6 +months on, power_model is best in toggle too. + +### Observation — profile spotlight (pooled over campaigns) + +| profile | method | bias | RMSE | max |error| | +|---|---|---|---|---| +| prepost `rated_plus_5pct` | **power_model** | **−0.38** | **0.62** | **1.05** | +| prepost `rated_plus_5pct` | v0_binned | −1.24 | 1.42 | 2.23 | +| toggle `rated_plus_5pct` | **power_model** | +0.16 | **0.25** | **0.49** | +| toggle `rated_plus_5pct` | v0_binned | −0.63 | 0.68 | 1.10 | + +power_model is **flat across all seven profiles** (prepost RMSE 0.61–0.63, bias ≈ −0.38 on every +one), whereas **v0 has a specific weak spot on the rated-power uprate** — its worst profile in both +modes (the only profile where v0's toggle RMSE, 0.68, is more than ~2× its others). A rated-power +change shifts power at high wind speeds where v0's binned power-curve has sparse, noisy bins; +power_model's continuous reference-conditioned fit has no such blind spot. On the placebo +(`cp_0pct`) all three are well-behaved (toggle v0 even edges power_model, 0.18 vs 0.24 RMSE). + +### Interpretation +- **Prepost: power_model is the better method, decisively.** Lower bias (−0.39 vs −0.73 pp), lower + spread (0.49 vs 0.72), ~40% lower RMSE than v0, and it wins the majority of cases head-to-head — + the F1 contrast lever (expected power through the references) cancelling the common-mode drift + that v0 corrects only through its detrend step. naive is not in contention (covariate shift). +- **Toggle: a near-tie that tips to power_model with data.** Naive and power_model are + near-identical and both beat v0 overall on RMSE; v0's larger spread and its failure to improve + with longer toggling are the cost of its binning. v0's only advantage is the 3-month toggle + campaign, where power_model carries slightly more bias (+0.37) before its variance collapses. +- **v0's rated-power weakness is the clearest single result.** It is the one regime where v0 is + both biased and high-variance in *both* modes, and where power_model's flatness is most valuable. + +### Implications +1. **Answers F3's open question: power_model ≥ v0 in both modes for P50** — strictly better in + prepost and at toggle ≥ 6 months, with v0 ahead only at the shortest toggle campaign. It is now + the baseline to beat (G-level), not just vs naive. +2. **Short-toggle bias is power_model's one soft spot** — the +0.37 pp at 3-month toggle is the + thing to chip at next (mirrors the residual prepost bias noted in F3 #3); candidates are the + baseline-horizon / recency weighting already on the list. +3. **Add a rated-power-uprate case to any v0 regression framing** — it is v0's worst regime and a + natural demonstrator for power_model's advantage; worth a dedicated diagnostic. +4. Reproduce/extend with `study_power_model_compare.py` (`--skip-run` to re-merge, `--modes` to + restrict); it reuses the frozen slow v0 so each power_model iteration is cheap. + +--- + +## F3 — A simple counterfactual power model halves prepost bias and spread vs naive; toggle is a wash + +*2026-06-29 — new method `power_model` (the simplest-possible ML method: a single LightGBM +counterfactual power model). Source: `benchmarking/baselines/example_{prepost,toggle}_study.py` +run over the four +`example_profiles` (`constant_cp`, `wind_speed_cp`, `ti_cp`, `rated_power`), `n_replicates=4`, +campaign sweep 3/6/9/12 months, seed 0, scoring **naive + power_model** (oracle anchor; v0 off). +256 runs (64 per mode per method); the oracle's max |signed error| was 0.0, confirming harness +wiring. Single-case cross-check: `benchmarking/baselines/inspect_prepost_hard_case.py`.* + +`power_model` fits the test turbine's power on curated reference-only features (each reference's +active power + availability, all raw ERA5 columns) over the baseline and predicts the +counterfactual over the upgraded window: `uplift = sum(actual)/sum(counterfactual) − 1`. It is the +contrast lever F1 recommended (expected power expressed *through* the references), now realised by +a much simpler estimator than the R-learner — no propensity, no cross-fit. + +### Observation — bias (mean signed error) and spread (std signed error), fractional uplift +Overall, by campaign type: + +| mode | method | bias | spread | MAE | +|---|---|---|---|---| +| prepost | naive_ratio | −0.63% | 1.20% | 0.97% | +| prepost | **power_model** | **−0.34%** | **0.54%** | **0.53%** | +| toggle | naive_ratio | +0.15% | 0.22% | 0.21% | +| toggle | power_model | +0.16% | 0.18% | 0.20% | + +By campaign length (the short-campaign story, serves G2): + +| mode | method | 3mo | 6mo | 9mo | 12mo | +|---|---|---|---|---|---| +| prepost | naive_ratio | −1.08% / 1.65% | −0.60 / 1.19 | −0.38 / 0.95 | −0.45 / 0.81 | +| prepost | **power_model** | **−0.69% / 0.54%** | −0.23 / 0.51 | −0.23 / 0.55 | −0.21 / 0.43 | +| toggle | naive_ratio | +0.31% / 0.22% | +0.16 / 0.23 | +0.06 / 0.21 | +0.08 / 0.12 | +| toggle | power_model | +0.36% / 0.08% | +0.12 / 0.18 | +0.10 / 0.17 | +0.05 / 0.10 | + +(bias / spread per cell). Both methods are **flat across the four upgrade types** — in prepost +power_model holds bias ≈ −0.34% and spread ≈ 0.55% on *every* profile (naive ≈ −0.6% / 1.2% on +every profile), so neither method has a profile-specific weak spot. + +Single hard case (F1's placebo: `cp_0pct`, `T07`, 6-month prepost, truth 0%): power_model reads +**−0.18%** where the cross-fit R-learner was ~−14% biased (naive +0.58%, v0 ~0%). + +### Interpretation +- **Prepost is the clear win.** power_model roughly **halves both bias and spread** vs naive + (spread 1.20%→0.54%, MAE 0.97%→0.53%), and its edge is largest at short campaigns — 3-month + prepost spread 0.54% vs naive's 1.65% (~3× tighter), degrading far more gracefully as data + shrinks. Expressing expected power through the references cancels the common-mode seasonal/ + long-term drift that naive's raw pre/post ratio carries — the F1 contrast lever, delivered by a + method far simpler than the R-learner (which *amplified* prepost error instead). +- **Toggle is a wash.** Both methods are near-unbiased with ~0.2% spread. On interleaved on/off + blocks there is little covariate shift for the model to correct, so the ML adds nothing over the + naive ratio. power_model carries a touch more bias at 3-month toggle (+0.36%) but the tightest + spread (0.08%). + +### Implications +1. **power_model is the new baseline to beat** for prepost, and a credible drop-in for toggle. + It is wired into both example study drivers and `inspect_prepost_hard_case.py` in place of the + R-learner (which is kept as a comparator but no longer invested in). +2. **This is vs naive, not yet vs v0.** v0 was excluded for speed. The honest open question + (G-level) is whether power_model beats v0 in *both* modes — v0 held prepost bias to ~−0.006 to + −0.014 (F1), comparable to power_model's −0.0034 here, so a direct v0 vs power_model prepost + run is the next comparison. Toggle is where neither naive nor the R-learner beat v0, so a v0 + toggle comparison is the priority there. +3. The small residual prepost bias (~−0.3%, slightly negative at every length) is worth a look — + plausibly mild non-stationarity or filter asymmetry between the long baseline and the post + season; a candidate for the future season-matched / recency-weighted baseline horizon. + +--- + +## F2 — The prepost R-learner bias is driven by reactive-power and pitch reference features acting as calendar-time proxies + +*2026-06-26 — Issue 5 (cross-fit R-learner). Source: the ablation driver +`benchmarking/baselines/inspect_prepost_feature_ablation.py`, which pins the F1 hard case +(`cp_0pct` placebo on `T07`, 6-month prepost, true uplift 0%) and the identical `MethodInput`, +then re-runs the R-learner with reference features removed before feature-building. Follows up F1.* + +### Observation +On the F1 placebo case, dropping the reference **reactive-power** feature, then the **pitch** +features as well, removes almost all of the bias (truth = 0%, fixed seed, only the feature set +changes between arms): + +| arm | dropped tags | R-learner estimate | error | +|---|---|---|---| +| full (all features) | — | **−22.26%** | −22.26% | +| no reactive power | `wtc_ReactPwr_mean` | **−5.81%** | −5.81% | +| no reactive power, no pitch | `wtc_ReactPwr_mean` + `wtc_PitcPos{A,B,C}_mean` | **−1.39%** | −1.39% | + +Reactive power alone accounts for ~16 of the ~22 points of bias; adding pitch removes most of the +rest, leaving the placebo within ~1.4% of zero. + +*(The full-feature estimate is −22.3% here vs the ~−14% quoted for this case in +`inspect_prepost_hard_case`'s docstring. The docstring predates the `mandatory availability filter` +commit `f8c3491`, which changed row selection; the cross-arm comparison is internally consistent +regardless of the absolute level.)* + +### Interpretation — direct evidence for F1's overlap-failure root cause +F1 attributes the prepost bias to an overlap/positivity failure: the propensity model reconstructs +"is this the upgraded season?" from seasonally/temporally varying reference features. This ablation +localises *which* features carry that signal. Reactive power and pitch are the top propensity +features (F1 diagnostics), and the reactive-power diagnostic plots show its **control regime changes +over calendar time** — so it is a near-deterministic clock. Remove that clock and the propensity +model can no longer separate the long baseline from the upgraded season, so the `t_res → 0` +blow-up and the confounding it drives both shrink. This is the F1 root-cause #1 mechanism shown +end-to-end, with reactive power identified as the dominant temporal proxy and pitch as secondary. + +**Caveat:** this is a diagnosis, not a fix. Dropping informative features only removes the proxy +*channel*; any reference feature with a time trend (or a post-treatment correlation) can re-open it, +and discarding genuinely predictive signal is the wrong long-term lever. The principled fixes remain +F1's: a test-vs-reference contrast (so common-mode temporal drift cancels before the ML sees it) +and/or a season-matched baseline window. + +### To pick up Monday +1. **Isolate pitch** — run the missing arm (drop pitch only, keep reactive) to split their + contributions cleanly; the current run only brackets them. +2. **Confirm it generalises** — repeat across the other hard cases / a couple of seeds / the + non-placebo profiles (does removing the proxies also tame the +76% `cp_plus_10pct` overshoot?). +3. **Watch the propensity diagnostic** — re-check `propensity_std` per arm; the hypothesis predicts + it falls back toward the flat base rate as the temporal proxies are removed. +4. **Decide the lever** — feed this into the F1 direction choice: feature hygiene/guarding vs the + test-vs-reference contrast. The contrast is still expected to be the larger, more principled win. + +--- + +## F1 — The R-learner is accurate in toggle but fails in prepost (overlap/confounding) + +*2026-06-26 — Issue 5 (cross-fit R-learner). Source: the overnight studies +`benchmarking/baselines/study_overnight_{toggle,prepost}.py` over the seven shared +`overnight_profiles`, `n_replicates=4`, campaign sweep 3/6/(9)/12 months, incl. v0.* + +### Observation +In **toggle** the R-learner is excellent — it tracks the oracle and matches or beats v0. In +**prepost** it is badly biased, and the error grows with the magnitude of the true effect (it +amplifies). Mean P50 estimate vs truth, 3-month campaign: + +| profile | truth | toggle rlearner | prepost rlearner | +|---|---|---|---| +| cp_0pct (placebo) | 0.000 | +0.003 | **−0.158** | +| cp_plus_3pct | 0.020 | 0.023 | **−0.010** | +| cp_plus_10pct | 0.068 | 0.072 | **+0.763** | +| cp_minus_10pct | −0.068 | −0.066 | **−0.359** | +| ws_dependent_cp | 0.033 | 0.036 | +0.123 | + +Toggle bias is ~0.001–0.003 with tiny spread; prepost overshoots (+10% → +76%, −10% → −36%) and +even the 0% placebo reads −16%. Longer campaigns shrink but do not fix it (cp_plus_10pct prepost: +0.76 → 0.33 → 0.19 at 3/6/12 months). By contrast v0 holds prepost bias to ~−0.006 to −0.014. + +### Evidence — the propensity diagnostic +From the per-run `results` / `data_stats` CSVs (same test turbine, same upgrade window): + +- **Toggle:** `propensity_mean ≈ 0.500`, `propensity_std ≈ 0.11`. Baseline and upgraded both span + the same dates (interleaved 20-on/20-off), so they share a weather distribution. Propensity ≈ 0.5 + everywhere → the treatment residual `t − e_hat` is healthy → the R-learner collapses to clean + regression adjustment. This is the regime it is designed for (design note §4). +- **Prepost:** `propensity_mean ≈ 0.11`, `propensity_std ≈ 0.27`. The `data_stats` show why: + **baseline ≈ 2 years (2016→2018), upgraded ≈ one 3-month season (early 2018)**. Treatment is a + deterministic function of calendar time, against a long, season-mismatched baseline. + +### Root cause +This is an identifiability problem, not a bug. There are no timestamp features (by design — shuffled +K-fold cross-fitting assumes it), so in prepost the only contrast available is across time. That +breaks the estimator two ways, both visible in the diagnostics: + +1. **Overlap / positivity failure.** The propensity model partially reconstructs "is this the + upgraded season?" from seasonally-varying reference features (`std 0.27`, far from the flat 0.11 + base rate). Where `e_hat` drifts toward the upgraded window, `t_res → 0`, the pseudo-outcome + `y_res / t_res` blows up, the `t_res²` weights concentrate on a few high-leverage rows, and the + effect model extrapolates — producing the variance explosion and the magnitude-scaling bias. +2. **Unmodelled non-stationarity → confounding.** The outcome model `m(x) = E[Y|X]` is pooled over + the long baseline but never forms a *test-vs-reference contrast*. Any drift in the test turbine's + power relative to the references over those two years (seasonal `Y|X` shifts, air density, icing, + direction/wake differences between a winter-heavy baseline and the specific post season) lands in + the post-window residual and, because treatment ≡ time, is attributed straight to `tau`. The + placebo −16% is pure confounding with no real effect. + +v0 survives prepost precisely because it differences the test/reference power ratio and detrends, +cancelling the common-mode drift the R-learner leaves in. + +### Implications / candidate directions (not yet actioned) +In rough order of expected leverage: + +1. **Give the R-learner a test-vs-reference contrast** (the v0 lever): model the test/reference + power *ratio* (or difference) as the outcome so common-mode seasonal/long-term drift cancels + before the ML sees it. Largest expected win for prepost. +2. **Match the baseline window to the post window** (same season, comparable length) rather than + pooling the full multi-year history — reduces non-stationarity, though it does not restore + within-`X` overlap. +3. **Treat the R-learner as a randomised-treatment (toggle) method.** Consistent with the design + note framing (it collapses to regression adjustment when the propensity is flat), the honest + Phase-1 conclusion may be: R-learner for toggle, v0/reference-ratio for prepost — with the + prepost overlap failure documented as a finding. + +Relative to Issue 5's "done when" (P50 similar/better than v0 in both prepost and toggle): **met for +toggle, not met for prepost.** + +### Secondary observations from these runs +- **Stale outputs intermixed.** The earlier output directories carried `leaderboard_all_profiles.csv` + and several per-profile files from an older study (no rlearner rows, old profile names, a 9-month + grid) alongside the fresh overnight outputs. Addressed by writing each run to a fresh timestamped + folder with a `run.log` recording the git commit and study config + (`benchmarking/baselines/overnight_common.py`, wired into both overnight scripts). +- **Both overnight runs were truncated** (ran out of wall-clock on the slow v0 step, not a crash): + prepost completed 5 of 7 profiles, toggle 6 of 7. With `include_v0=True`, v0 dominates the budget + (~hours per profile vs seconds for rlearner/naive) — see the "skip v0 in initial passes" note. diff --git a/docs/v1/goals.md b/docs/v1/goals.md new file mode 100644 index 00000000..87e7a3f6 --- /dev/null +++ b/docs/v1/goals.md @@ -0,0 +1,91 @@ +# wind-up v1 — goals + +## Context + +wind-up v0 is a beta tool (published on PyPI as `res-wind-up`) that measures the +energy-yield **uplift** of a turbine upgrade — `(upgraded MWh) / (baseline MWh) − 1` +for matched conditions — using a **binned power-curve, test-vs-reference** method +with uncertainty estimation (including a block bootstrap). It is used in production +at RES. + +v1 is a major upgrade. Its purpose is to make wind-up measure uplift **more +accurately**, from **shorter campaigns**, and with **more insight into how an +upgrade's performance depends on conditions** — while keeping the tool practical +and easy to use. + +## North-star vision + +> Turn wind-up from a single-method tool into a **platform for measuring +> turbine-upgrade uplift**, in which alternative methods are pluggable and +> objectively benchmarked, and in which more of the final report is produced by +> the tool rather than by hand. + +## Goals + +### G1 — Pluggable, objectively benchmarked methodology *(the central thrust)* +The choice of "best" uplift method is currently unknown and should not be +constrained by premature architectural assumptions. v1 makes the uplift method a +**pluggable component** and provides a **public synthetic-data evaluation harness** +with known ground truth, so candidate methods can be compared objectively against +the v0 baseline. New methods are merged only once they demonstrably beat the +baseline. + +### G2 — Better results from shorter campaigns +Reduce the data (campaign duration) needed to reach a given accuracy/precision. +Short-campaign robustness is a primary evaluation axis, not an afterthought. + +### G3 — Conditional / heterogeneous uplift information +Report not just a single uplift number but **how uplift varies with conditions** — +wakes vs free-stream, day vs night, wind direction, atmospheric stability — so +upgrades can be understood and targeted. (This is a natural output of +treatment-effect / ML methods conditioned on more than wind speed.) + +### G4 — Pipeline as independent, composable steps +Make each stage of the analysis runnable on its own: +1. **Pre-processing** — source data (SCADA, ERA5, optionally mast/LiDAR), filter, + feature engineering. +2. **Measure campaign uplift + uncertainty** per turbine. +3. **Long-term extrapolation** of uplift + uncertainty. +4. **Aggregate** results across turbines. + +### G5 — Matured I/O and configuration +Make wind-up easier to configure and use (cleaner inputs/outputs, clearer config). + +### G6 — More of the report auto-generated +Today wind-up produces most of the per-pair analysis plots (the report appendix), +but the executive summary, results tables, farm-layout figure, exclusion-period +table, combined/by-pair uplift charts, and reference-suitability table are made by +hand. v1 should generate **more of this report content** to cut manual +post-processing. *(Lower priority than G1–G4; high practical value.)* + +## What v1 is NOT (initial scope guards) + +- **Not** an uncertainty-model overhaul *first*. The first methodology effort + targets **P50 accuracy and precision only**. The P95 / uncertainty model + (block bootstrap, conformal OOD, density-ratio long-term weighting) is + **deferred** until the best P50 method is identified, because P95 depends on the + chosen point method. +- **Not** a big-bang rewrite. Foundational refactors (G4, G5) are done only as far + as needed to unblock the methodology work — enough to avoid baking in + assumptions, no more. + +## Success criteria + +1. A candidate method recovers a **known injected uplift** on synthetic datasets + more accurately and/or precisely than the v0 baseline (P50). +2. A method reaches a target accuracy/precision from a **shorter campaign** than + the v0 baseline requires. +3. The uplift method is selectable via config, with the v0 method preserved. +4. (Stretch / later) The conditional-uplift story (G3) and auto-generated report + content (G6) are available from the tool. + +## Source material + +- Internal v1 planning notes — the v1 plan and goals. +- `wind-up-ml-uplift-design-note.md` — the ML / treatment-effect (R-learner) + methodology candidate and staged plan. +- Example v0 uplift reports (Hill of Towie AeroUp/Pitch, Tallentire, Earlseat, + wake-steering) — basis for the report-generation gap analysis (G6). +- Key open data: Hill of Towie SCADA (Zenodo 20204946) and + `resgroup/hill-of-towie-open-source-analysis`. +- Other open data: SMARTEOLE (see existing wind-up example notebook), Kelmarsh (https://zenodo.org/records/16807551), Penmanshiel (https://zenodo.org/records/16807304), WeDoWind (pitch-angle and vortex-generator examples in wind-up). diff --git a/docs/v1/issues.md b/docs/v1/issues.md new file mode 100644 index 00000000..3a8fd422 --- /dev/null +++ b/docs/v1/issues.md @@ -0,0 +1,877 @@ +# wind-up v1 — first issues (drafts) + +Drafts of the concrete issues, to be refined here and then created as GitHub +issues on `resgroup/wind-up`. Issues 1–8 cover **Phase 1** (see +[roadmap.md](roadmap.md)): the public evaluation harness, the v0 baseline, the +minimal data contract, and the first new candidate methods — all judged on **P50 +accuracy and precision only**. Issues 9+ are the **second wave**, drafted after +Issue 8 shipped: improving `power_model` (the current best method) on overall and +conditional P50, then extending the measurement to AEP uplift and starting the +uncertainty (Phase 3 / WS4) work. + +Suggested order of execution: #1 → #2 → #3 → #4 → #5 (done), then +#9 → #10 → #11 → #12 → #13 → #14 → #15 (done) → #16 → #17 → #18 → #19. (#9–#12 are +independent input-data and model trials sharing one evaluation protocol — many ideas, +each tested one by one; #14 is small and independent, so it can be pulled earlier — and +#15's adaptive rule wants #14's hardened conditional scoring in place; #16/#17 are the +`power_model` hygiene + consistency follow-ups that fall out of #15 and are best done +before the AEP/uncertainty work builds on the conditional path; #19 builds on #18's AEP +machinery, so it goes last.) (Issue 4 was originally a data contract + method-selector +issue; it has been re-scoped to a naive energy-ratio method — see Issue 4 below for why.) + +--- + +## Issue 1 — Synthetic upgrade-dataset generator (WS1) + +**Goal:** generate realistic SCADA-like datasets with a *known* injected uplift, to +serve as ground truth for evaluating uplift methods. + +**Scope** +- Select real SCADA from stable, no-upgrade periods/turbines (start with Hill of + Towie open data; design for other open wind farms). +- Inject known uplift *profiles*: + - constant Cp change; this increases or decreases power in region 2 but does not change rated power + - wind-speed-dependent Cp change; + - condition-dependent (e.g. direction/wake-dependent — wake-steering shape, or stability dependent as TuneUp may be) Cp change. + - rated power change +- Preserve realistic structure (autocorrelation, reference turbines, conditions) + so methods can't trivially detect the injection. +- Emit the dataset plus a machine-readable record of the true injected uplift + (overall and per-condition) for scoring. + +**Done when:** a documented function/CLI produces ≥1 synthetic dataset per profile +from open data, with the ground-truth uplift recorded alongside. + +--- + +## Issue 2 — P50 evaluation harness & scoring (WS1) + +**Goal:** score any uplift method's P50 estimate against the known injected truth, +including a short-campaign robustness sweep. + +**Scope** +- Accuracy (bias) and precision (spread) metrics: recovered uplift vs injected, + overall and (where applicable) per condition. +- Short-campaign sweep: re-score as a function of campaign length / data volume to + quantify how accuracy and precision degrade with less data (serves G2). +- A simple results format / leaderboard so methods can be compared side by side. +- **P50 only** — no uncertainty/P95 scoring in this phase. + +**Done when:** given a method (conforming to the Issue 4 contract) and a synthetic +dataset, the harness emits comparable accuracy/precision numbers and a +campaign-length curve. + +--- + +## Issue 3 — Wire the v0 binned method as the baseline (WS2) + +**Goal:** run the existing v0 binned power-curve uplift method through the harness +to establish the bar every new method must beat. + +**Scope** +- Adapt the current pre/post power-performance pipeline to consume the Issue 4 + data contract and emit a P50 estimate the harness can score. +- Record baseline accuracy/precision and the short-campaign curve for each + synthetic profile. + +**Done when:** baseline P50 scores exist for all synthetic profiles and are the +reference point in the leaderboard. + +--- + +## Issue 4 — Naive energy-ratio method (WS3) + +**Goal:** a second, deliberately simple, fully independent method that validates the +harness is not implicitly tuned to v0 — and proves the existing thin method seam is +genuinely pluggable. + +**Why this, not the original "data contract" issue:** the thin +`MethodInput`/`MethodOutput` seam from Issues 2–3 already *is* the shared, method- +agnostic contract; the drafted "per test-reference conditioned dataset" was over-fit to +v0 (an R-learner fits once per test turbine over all references at once), and the +`assessment_method` production selector only earns its place once there is a winner to +promote. The durable kernel of the old issue — a treatment-invariant reference-only +feature builder + the §8 bias-guard test (design note §3/§8) — folds into Issue 5. + +**The method.** For a set of rows let `ρ = Σ test_power / Σ reference_total_power` over +*complete-case* timestamps (test turbine **and every** reference finite). Estimate +`uplift = ρ(treated) / ρ(baseline) − 1`. It never reads the test turbine's own wind +speed (design note §3), shares no code with v0, and has no wind_up dependency. It makes +no covariate-shift correction by design, so it is the "don't condition at all" floor: +biased on prepost, near-unbiased on toggle (interleaved on/off share a wind +distribution). + +**Scope** +- `NaiveRatioMethod` behind the existing `Method` seam; prepost **and** toggle. +- Rich per-run diagnostics (a data-stats CSV per `all`/`baseline`/`upgraded` segment, a + headline-results CSV, optional plots) so a human can confirm the right data was + received and interpreted; the headline uplift is re-derivable from the stats CSV. +- Add toggle support to `V0BinnedMethod` (wiring wind_up's native toggle assessment) so + v0 can be scored on toggle campaigns too. +- Add the naive method to the existing prepost driver; add a new toggle example driver + (3% Cp increase, 20-min-on/20-min-off) scoring naive + v0 + oracle. + +**Done when:** `naive_ratio` is scored alongside `v0_binned` and the oracle on the +synthetic profiles for both prepost and toggle, the per-run diagnostics are written, and +its accuracy/precision appears in the leaderboard. + +--- + +## Issue 5 — First candidate: cross-fit R-learner (P50) (WS2) + +**Goal:** implement the design-note R-learner as the first new method and score it against the baselines (naive and v0). + +**Scope** +- Cross-fit R-learner producing a P50 uplift, per the design note + (LightGBM outcome [L2/Huber] + propensity nuisances; effect model on residuals). +- Be sure to provide rich csvs and diagnostic plots; look at Naive for inspiration and grow from there depending on the method details +- There should be no restrictions / limitations on what data is provided (v0 dependency removed from harness in previous PR). To start with just provide the same data columns that v0 gets to prove the method is superior; even more data columns (eg all Min and Max fields) can be provided later +- Provide ERA5 data to the method as well. It will need to be upsampled to 10min timebase. Use a wind speed correlation sweep to sync with the SCADA (see existing wind-up code for inspiration) +- **Treatment-invariant reference-only features** — enforced; include a regression + test on a deliberately treatment-corrupted nacelle wind speed proving the + reference-only rule removes the post-treatment bias (design note §8). Reference + turbines affected by the wakes of upgraded turbines may need to be excluded in + certain wind directions (very relevant for wake steering). +- Another known feature to avoid is voltage (at the turbine's external connection); if the turbines are wired in series then the voltage drop across the test turbine will be approximately proportional to its active power, so if the method can see the voltage at the two turbines either side of the test turbine then estimating power is trivial and the power estimate is not using information about the weather. The giveaway that this issue is happening is feature importance; make sure feature importance diagnostic plots and log messages are emitted. +- Score on the harness across all synthetic profiles and the short-campaign sweep; + compare to naive first (fast), then the v0 baseline after naive is beaten. +- Uncertainty/P95 explicitly **out of scope** here (Phase 3 / WS4) but keep it in mind. +- Reporting uplift by condition (eg uplift by wind speed, direction, etc) out of scope for now but keep it in mind + +**Done when:** the R-learner runs through a prepost and toggle study and its P50 accuracy/precision vs the baseline is recorded in the leaderboard and similar or better than v0. + +--- + +## Issue 6 — Clip power_model predicted power to a sane range (power_model) + +**Goal:** stop the boosted counterfactual over/under-shooting the physically plausible +range at the extremes — a small precision gain, especially in the fragile tail bins. Do +this first: it is low-risk and serves as the simple test case for Issue 7. + +**Scope** +- In `PowerModelMethod._fit_predict` (`benchmarking/baselines/power_model/method.py`), + clip every model prediction (the upgraded counterfactual *and* the baseline-holdout + prediction) to `[lower, upper]` with `lower = min(0, min(y_train))` and + `upper = max(rated_power_kw, max(y_train))`, where `y_train` is the fitted-on baseline + outcome. (Tree boosting sums trees, so predictions can slightly exceed the training + `y` range; the clip binds only at the extremes — expect a small effect.) +- Add an optional `rated_power_kw: float | None = None` config field; when `None` the + upper bound is just `max(y_train)`. HoT rated = 2300 kW. +- Keep it a pure post-prediction transform so overall and conditional both use clipped + predictions consistently. + +**Done when:** predictions are bounded; existing recovery/placebo tests still pass; a +unit test on the clip helper confirms out-of-range predictions are pulled to the bounds +and in-range ones are untouched. + +--- + +## Issue 7 — Confirm the improvement-evaluation workflow surfaces what's needed (power_model) + +**Goal:** before the larger bias-correction work, prove that +`study_power_model_compare.py` gives the information needed to judge a power_model change +— using the Issue 6 clip as the first, simple test case. + +**Scope** +- Run `benchmarking/baselines/study_power_model_compare.py` (power_model vs frozen + v0/naive, 3-method plots) on the clipped power_model. +- Confirm it reports, side-by-side and legibly, both **overall** P50 error and the + **per-condition** (ws & TI) recovered-vs-truth curves for the `ti_dependent_cp` / + `ws_dependent_cp` hard cases — enough to see whether a change helped, hurt, or was + neutral, overall and per bin. +- If a needed view is missing (e.g. a corrected-vs-current conditional overlay against + truth, or a per-bin error table), add the minimal reporting to the study/inspect + scripts so Issue 8 can be evaluated the same way. + +**Done when:** a single command produces the overall + per-condition comparison for the +clip change, and the improvement is readable from it without ad-hoc digging. + +--- + +## Issue 8 — Cross-prediction bias-cancellation for shrinkage-driven conditional bias (power_model) + +**Goal:** remove the counterfactual model's **per-condition** (shrinkage) bias — the F5 root +cause — by cancelling it between two symmetric train/predict directions on weather-matched +data. The overall P50 is left as the single full-window fit (its whole-window shrinkage +integrates to ≈0, so it is already the cleanest headline); the two-direction correction is +spent on the per-condition decomposition, which is then re-leveled so its energy aggregation +equals that headline (F8). + +**Scope** +- **Matching-variable analysis (one-off, do first).** On Hill of Towie, run a feature- + importance analysis to choose which ERA5 variables to match on (likely wind speed + + direction, possibly more). Record the chosen set + rationale in `docs/v1/findings.md` + and hard-code it as the default matching set. ERA5 is preferred for its full coverage + and temporal stability; the set may later be tuned per wind farm. +- **ERA5 coarsened-exact matching (CEM).** New utility: bin baseline vs upgraded rows on + the chosen (synced) ERA5 variables; within each cell subsample the larger side to the + smaller side's count (seeded); drop one-sided cells. Yields equal-count, weather-matched + baseline/upgraded sets. The matching axis (ERA5) is distinct from the reporting/binning + axis (test-turbine ws/TI, kept as today so bins match ground truth). +- **Two directions + geometric combine.** Forward: train on matched baseline, predict + matched upgraded → `r_fwd` (overall and per bin via `energy_ratio_by_bin`). Reverse: + train on matched upgraded, predict matched baseline → `r_rev`. Combine + `uplift = sqrt((1+r_fwd)/(1+r_rev)) − 1` (exact under a common per-bin multiplicative + shrinkage); also emit implied bias `1/sqrt((1+r_fwd)(1+r_rev))` as a diagnostic. Guard + non-positive `(1+r)` and empty/sparse bins. +- **Default, not opt-in.** The matched two-direction cross-prediction is the **sole** + conditional-uplift method and is **on by default** via `conditional_uplift: bool = True` on + `PowerModelMethod` (an A/B `bias_correct` flag was used during development, then removed once + the approach won). It requires ERA5 (the matching axis) and runs **last** — nothing else + depends on it — so `conditional_uplift=False` skips the expensive cross-prediction and returns + only the overall P50. Per-run outputs: conditional CSVs in a `conditional/` subfolder, the + implied-shrinkage plot in `plots/7_conditional_uplift/`. +- Reuse the existing outcome model factory (`make_outcome_model`) and `CONDITION_BINS`; + diagnostics carry the implied-shrinkage `1/sqrt((1+r_fwd)(1+r_rev))` and the CEM balance, and + `study_power_model_compare.py` overlays the committed benchmark vs the current run vs truth per + covered `(profile, condition)` (a benchmark regression view). + +**Done when:** `study_power_model_compare.py` (Issue 7) shows the `ti_dependent_cp` / +`ws_dependent_cp` conditional curves materially flatter toward truth with the overall P50 +unchanged; a regression test recovers a known flat-zero placebo per-condition uplift to a small +absolute per-bin bias (the shrinkage tilt cancelled); findings.md updated. *Shipped 2026-07-03: +overall P50 bit-identical, conditional score roughly halved (prepost mean |bias| 18.2→6.3 pp, +toggle 13.0→4.1 pp) and the benchmark regenerated under the new default.* + +--- + +## Issue 9 — ERA5 feature engineering: derived quantities + hub-height interpolation (WS2) + +**Goal:** use ERA5 better. Today the model consumes the raw Open-Meteo columns (plus +direction sin/cos); derive the physically meaningful, §3-legal quantities that actually +drive turbine power and its scatter — including a turbulence proxy, which the model +currently lacks entirely (the F5 root cause: the §3 rule denies it the test turbine's own +ws/TI, so its residuals correlate with TI). This issue also establishes the shared +**candidate-by-candidate evaluation protocol** that Issues 10–11 reuse. + +**Evaluation protocol (shared by Issues 9–11).** There are many ideas for improving the +input data available to the model; they must be tested **one by one** (or in the smallest +sensible groups) so each one's effect is known. Per candidate: add it, A/B via +`study_power_model_compare.py` against the committed benchmark, and accept only what earns +its place. Gates: **overall P50 no worse** — the placebo prepost bias is the sharpest read +(a feature that imports temporal drift shows up there); **conditional** mean |bias| / +`implied_shrinkage` improved or neutral; **feature importance sane** (the F2 lesson: a +channel can act as a calendar proxy; the Issue 5 voltage lesson: importance diagnostics +are the giveaway). Record every accepted **and rejected** candidate with evidence in +`findings.md`; regenerate the benchmark JSON whenever a default changes. + +**Scope (the ERA5 candidates)** +- **Hub-height wind speed.** New optional `hub_height_m` config (Hill of Towie = 59 m). + Compute the local shear exponent `alpha = ln(ws_100m/ws_10m)/ln(10)` per row and + interpolate with the shear power law: `ws_hh = ws_100m · (hh/100)^alpha`. +- **Gust turbulence proxy:** `wind_gusts_10m / wind_speed_10m` — a unitless, TI-like + quantity (guard the calm-wind denominator). Do a comparison to real (SCADA) TI and play around with the calculation if it the ERA5 derived TI is not realistic. +- **Shear exponent** `alpha` as a feature in its own right (F6 already validated it as a + real independent signal and folded out the 10 m/100 m collinearity), **vertical veer** + (`wind_direction_100m − wind_direction_10m`, wrapped to ±180°), **air density** (from + temperature + surface pressure + humidity), and a **stability indicator** if the + available fields allow. +- Build as a **shared ERA5-derivation utility** so every method *and* the CEM matching + step reuse it (the shear derivation currently lives only in + `inspect_era5_matching_importance.py`). +- Optionally revisit `matching_vars` afterwards (e.g. the gust proxy or `alpha` in place + of raw gusts, per F6's deferral) — a separate benchmark regeneration if taken. + +**Done when:** the derivation utility exists with unit tests (known-input checks); every +candidate has an accept/reject verdict recorded in `findings.md`; the benchmark is +regenerated under the accepted default set with overall P50 and conditional no worse. + +--- + +## Issue 10 — Time features: campaign-relative drift, season, solar position (WS2) + +**Goal:** today the timestamp is dropped before modelling, so the model cannot know about +anything that varies with time but not weather. Trial explicitly-constructed time features +— the headline motivation is **reference turbines changing over time**, a major bias risk +and one of the main challenges the method must deal with: a drift feature gives the model +a chance to absorb reference change instead of attributing it to the upgrade. It is also possible to ERA5 to drift over time vs the site due to ERA5 input data changes over time and site exposure changes over time (eg forestry growth, new neighbour wind farm, etc.) + +**Scope (one by one, per the Issue 9 protocol)** +- **`time_since_campaign_start`** continuous value in unit of days (negative in the baseline). Lets the model track + slow reference change relative to the campaign. +- **Time of year:** continuous value measuring the Julian day offset to a meteorologically meaningful anchor (say + June 21), split into sin/cos components — tells the model roughly what season it is, + potentially useful beyond the instantaneous weather data. +- **Time of day, as solar altitude + azimuth** computed from an optional lat/long config + and the UTC timestamp (a physically meaningful encoding of diurnal cycle; calculation + code exists in other projects and can be ported). +- **Known caveats to test against, not reasons to skip.** In prepost, + `time_since_campaign_start` is treatment-collinear, and trees cannot extrapolate — for + upgraded rows the feature exceeds every training value, so predictions clamp at the + boundary leaves: it can encode baseline-internal drift but freezes it at the changeover. + Season features against a < 12-month baseline are partial calendar proxies. Whether each + helps or imports bias is exactly what the placebo gate decides — run it at multiple + campaign lengths in **both** modes before accepting. +- The R-learner's "no timestamp features" rule (later-work list) was about shuffled + cross-fitting; power_model has no cross-fitting, so trialling these is legitimate — but + a time feature dominating the importance ranking is a red flag. + +**Done when:** each time feature has an accept/reject verdict with placebo evidence in +both modes in `findings.md`; the benchmark is regenerated if any are accepted. + +--- + +## Issue 11 — Reference power statistics features (max / min / SD) (WS2) + +**Goal:** give the counterfactual a local variability/turbulence signal through the most +**calibration-stable channel** — the reference power signal. Bring in each reference's +active-power **max, min and SD** companion fields (present in the Hill of Towie open +data; usually available from any wind farm). Within-period power SD in particular is a +§3-legal turbulence proxy, sited at the farm rather than at ERA5's grid scale. + +**Why reference power and not reference wind speed:** reference nacelle wind speed / +wind-speed SD were considered and **rejected**. A reference anemometer is at high risk of +changing calibration over time; in a prepost campaign that drift would be read as uplift +(a toggle campaign might tolerate it, but prepost and toggle share one code path, so the +risky feature stays out of both). Reference *power* is more likely to be stable — and +since references are usually the same turbine type as the test turbine, exposed to the +same performance-degradation tendencies, it leads to a fairer expectation of what the +test turbine could have produced. + +**Scope** +- Wire the max/min/SD active-power fields through `build_reference_features` (same + `" @ "` naming; NaN-tolerant as today; `check_reference_only` still + applies). Column names configurable like the existing `active_power_col`. +- A/B per the Issue 9 protocol: one field (or the trio) at a time, placebo gate, + importance watch — power max/min/SD could in principle carry a drift signature too + (e.g. a reference derate changes its max), which is precisely what the placebo run + detects. +- Confirm the harness/synthetic path carries these columns unmodified for reference + turbines (only the test turbine's signals are injected). + +**Done when:** verdict per field recorded in `findings.md`; benchmark regenerated if +accepted; the rejected reference-anemometer alternative and its rationale are noted in +the findings entry. + +--- + +## Issue 12 — Outcome-model fundamentals: objective, hyperparameters, calibration slope, alternative learners (WS2) + +**Goal:** revisit the basics of the counterfactual model itself. Today it is a LightGBM +regressor with the L2 objective and fixed design-note hyperparameters (600 trees, +lr 0.03, 63 leaves, `min_child_samples=200`, subsample/colsample 0.8) — never tuned, no +early stopping, and identical whether the fit has ~13k training rows (3-month toggle) or +~250k (2-year prepost baseline). + +**Two framing principles (record them in findings.md so they aren't relitigated):** +- **The objective must target the conditional mean.** The estimand is an energy ratio and + energy is a sum of conditional means, so L2 stays the default (design note §2). Power + conditional on features is skewed (near cut-in and around rated), so median-type + objectives — MAE, Huber in its robust regime, quantile-0.5 — estimate the median and + would bias the energy sum. The F5 shrinkage is a *regularisation* artefact, not a loss + artefact; changing objective does not fix it. Legitimate within the mean family: + **Tweedie** or variance-weighted L2 for the strong heteroscedasticity of power + (an efficiency candidate, not a bias fix). +- **Tune on uplift metrics, never on prediction RMSE.** More regularisation can improve + held-out RMSE while *worsening* shrinkage — the two objectives disagree exactly where + it matters. Yardsticks: placebo bias on the harness, per-bin residual flatness and + predicted-vs-actual **calibration slope** (target ≈ 1) on a time-blocked held-out + baseline, and replicate spread. Guard against overfitting the benchmark: tune on the + held-out-baseline proxies, confirm on the harness, ideally on turbines/windows not used + for tuning. + +**Scope (per the Issue 9 protocol, one candidate at a time)** +- **Data-size-adaptive capacity:** early stopping on a time-blocked validation split; + scale `min_child_samples` / leaves with training size (the 3-month and 24-month fits + differ ~10× in rows but share one capacity today). +- **`linear_tree=True`:** piecewise-linear leaves reduce the flat-leaf compression at the + edges of the feature distribution — a direct attack on the F5 shrinkage (and it + softens the Issue 10 boundary-clamping caveat, since linear leaves extrapolate). +- **Post-hoc calibration-slope correction:** fit actual-vs-predicted on time-blocked + held-out baseline (linear, or isotonic), apply to the counterfactual predictions — the + cheap de-shrinking cousin of Issue 13's residual calibration; measure how much each + contributes when combined. +- **Seed ensembling:** average K seeds (subsample/colsample noise) for variance + reduction; measure the replicate-spread gain vs run-time cost. +- **Alternative learners behind a model-factory seam** (make the outcome model injectable + on `PowerModelMethod` rather than hard-coded to `make_outcome_model`): CatBoost / + XGBoost as same-family sanity checks; a small tabular NN or TabPFN for the short- + campaign regime; and a deliberately low-variance structured baseline (GAM / linear on + hub-height ws + direction features from Issue 9) — lightly regularised so nearly + shrinkage-free, valuable as a cross-check on the tree models' conditional bias even if + its overall accuracy is worse. +- **Quantile objectives are out of scope for the point estimate** (median ≠ mean under + skew → biased energy). Quantile models return in Issue 19 / WS4 for conformal OOD + filtering and diagnostics — not as the uplift estimator. + +**Done when:** a findings entry records the verdict per candidate against the uplift +yardsticks; defaults change only where those improve; the benchmark is regenerated if +defaults change. + +--- + +## Issue 13 — Calibrate out the structural headline bias; revisit toggle's campaign-only training (WS2) + +**Goal:** remove the persistent overall-P50 bias: prepost ≈ **−0.4 pp on every profile** +(F3/F4) and toggle ≈ +0.4 pp at 3 months. The flatness across profiles says it is a model +artefact, not effect-dependent — but the two modes get there by different mechanisms, and +this issue addresses both. **Prepost:** the model's conditional bias `b(x)` integrates to +≈ 0 over the *training* (baseline) covariate mix, but the headline evaluates it over the +*upgraded* window's mix — any weather/seasonal shift between the windows converts +conditional bias into headline bias. Estimate that term on untreated data and subtract it +(this is F5's implication #1, baseline-residual calibration, never actioned — aimed at +the headline rather than the bins, which Issue 8 already fixed). **Toggle:** under +`toggle_campaign_only=True` the train (off) and predict (on) rows interleave through the +same weeks, so train/predict covariate shift is minimal by construction — the 3-month +bias is more plausibly small-sample shrinkage from the tiny fit (~6–7k off rows vs ~250k +for a 2-year prepost baseline). The candidate fix is more training data: the pre-campaign +baseline that `toggle_campaign_only` currently drops, made safe by exactly this issue's +calibration plus Issue 10's time features. + +**Scope** +- **Time-blocked baseline cross-validation.** Replace the random 20% holdout in + `_holdout_fit` with contiguous time blocks (out-of-fold prediction for every baseline + row). The current shuffled split leaks autocorrelation (holdout rows sit minutes from + training rows), so its residuals are optimistic — fine as a display, unusable as a + calibration basis. This also makes the step-5 fit-quality diagnostics honest. +- **The calibration.** From the out-of-fold baseline residuals, estimate the mean residual + per ERA5 cell (reuse the CEM coarsening); weight cells by the *upgraded* window's + occupancy to get the expected headline bias under the upgraded mix; subtract it from + `Σcounterfactual` (guard cells unseen in baseline — fall back to the global mean + residual). For toggle, the campaign's off rows are untreated data under (almost + exactly) the on rows' covariate mix, so their out-of-fold residuals estimate the + headline bias more directly than the ERA5-cell reweighting prepost needs. +- **Toggle training-window revisit.** Today `toggle_campaign_only=True` drops every + pre-campaign row, so a 3-month toggle fits on the campaign's off rows only. Trial + including the pre-campaign baseline as a 2×2 on the placebo sweep — {campaign-only, + all-data} × {calibration off, on} — with the Issue 10 time features in place. + `time_since_campaign_start` is *not* treatment-collinear in toggle (on and off + interleave), so the model can learn baseline→campaign drift and interpolate within the + campaign, largely voiding the Issue 10 boundary-clamping caveat. Win condition: the + extra rows cut shrinkage/spread without importing drift bias. Decide the + `toggle_campaign_only` default from that evidence (`naive_ratio` keeps campaign-only + regardless — there the restriction *is* the method's distribution matching). +- **Time-decay weights are a contingency, not a deliverable.** If the placebo with + pre-campaign data shows drift-driven bias that the time features and the calibration do + not absorb, trial exponential time-decay sample weights before rejecting pre-campaign + data outright; otherwise the age-of-data feature carries the recency signal and keeps + full effective sample size. +- **A/B behind a flag** (the Issue 8 playbook): placebo-centred evaluation across the + campaign sweep in **both modes** — the win condition is placebo bias → ≈ 0 **without** + a spread cost at short campaigns (the F7 failure mode to avoid). +- **Sequencing with Issues 9–12:** run after (or with) the feature and model trials — + better features and a better-calibrated model shrink `b(x)` and therefore the + correction (Issue 12's calibration-slope fix is the closest cousin); report how much + bias remains for this calibration once those land. +- Keep the Issue 8 conditional path consistent: the re-level target becomes the + calibrated headline. + +**Done when:** pooled prepost |bias| materially reduced (target ≲ 0.1–0.2 pp on the +placebo) at no spread cost at any campaign length; the 3-month toggle bias explained, and +either materially reduced or the campaign-only default re-confirmed with placebo +evidence; a recorded decision on the `toggle_campaign_only` default; findings entry +quantifying both mechanisms; benchmark regenerated if the calibration or the toggle +default changes. + +--- + +## Issue 14 — Harden the conditional decomposition: count floor, imputation, balance, re-level coverage (WS2) + +**Goal:** fix the three known remaining defects of the two-direction conditional +estimate, so every reported per-bin number is either trustworthy or an explicitly +flagged, physics-informed fallback — and keep the scoring honest about the difference. + +**Motivation strengthened by Issues 12–13 (F14–F16):** three further independent hits on +the un-floored sparse tail. (a) F15's residual calibration was partly sunk by sparse-cell +noise (~13 rows/cell on a 3-month toggle). (b) F16: merely *sample-weighting* the matched +direction fits flipped degenerate extreme-TI bins to three-digit per-bin uplift (+1039 % +in one replicate) at some decay half-lives and not others — replicate chance, i.e. the +tail bins sit on a knife edge; the time-decay weights had to be confined to the headline +fit as a workaround, and the floor should make the conditional path robust enough to lift +that restriction. (c) Throughout the F14–F16 A/B screens the conditional *mean* deltas +were dominated by the same few exploding tail cells, forcing verdicts to be read from +overall rows and cell tallies — the floor is also what makes the conditional benchmark +signal itself trustworthy. + +**Scope** +- **Per-reporting-bin matched-count floor** (the F7/F9 follow-up): below a minimum + matched count per side in a reporting bin, replace the overshooting two-direction + combine with the imputed value (next bullet) and set `covered=False` in the + conditional CSV — a flagged fallback, never a bare NaN, so every bin stays scoreable + and abstention can't game the leaderboard (`summarize_errors` drops non-finite errors, + so NaN-ing hard bins would otherwise improve the conditional score for free). The + floor is on *effective* per-side counts — Kish ESS `(Σw)²/Σw²` if the balance + reweighting below is adopted, raw counts otherwise — with the threshold chosen on + placebo/benchmark evidence across many bins, not tuned to the one known bad TI bin. + Kills the sparse-extreme swings (e.g. TI (0.45,0.50] −77 → +93 pp). +- **Physics-informed imputation for floored ws bins.** Uplift vs wind speed has a known + shape for most upgrades (Cp-maximising from cut-in to rated, rated power thereafter): + fill uncovered low-ws bins from the closest covered bin above (bfill on ascending ws), + then fill everything above the last covered bin with **0 uplift** (at rated, baseline + and upgraded both hit rated power). Two documented caveats: the 0-at-rated prior is + wrong for uprating/power-boost upgrades (keep the imputer a documented, replaceable + default and let benchmark profiles that violate it expose it), and when coverage stops + well below rated the 0-fill is conservative for the in-between bins. TI has no such + ordering physics — impute floored TI bins at the overall uplift. +- **Re-level with imputed bins pinned.** `_relevel_conditional` computes the aggregation + over covered (finite-shape) bins only, but the target `1+overall` includes the energy + of *uncovered* bins — so covered bins absorb the uncovered bins' MWh and λ is tilted + when coverage is imperfect. Fix: hold imputed bins fixed at their imputed uplift + (pinned, not λ-rescaled) and solve λ over the measured bins only, so measured + + imputed together satisfy the headline identity exactly (guard the degenerate case of + no measured bins → overall-only). With the floor creating more imputed bins, this + matters more than today. +- **Per-reporting-bin balance.** The shrinkage-cancellation premise is *per bin*, but CEM + equalises the ERA5-cell mix only globally — within one ws/TI reporting bin the forward + and reverse row sets can still have different weather mixes (partly because the bin + axes are the test turbine's post-treatment nacelle signals, so the same nominal bin + samples different weather pre vs post). Post-stratify within each reporting bin to the + **intersection** of the two directions' ERA5-cell supports (no new model fits needed): + honest but thins sparse bins, which is fine — the ESS floor then catches them instead + of a heavily-reweighted number sneaking through. Measure whether it moves the per-bin + bias before adopting. +- **Coverage in the scoring.** Surface coverage next to conditional accuracy in the + leaderboard/comparison outputs (e.g. fraction of upgraded energy in `covered` bins), + and compare A/B variants on commonly-covered bins as well as overall — imputed bins + are scored like any other (the flag distinguishes them), so the imputation prior is + itself benchmarked. +**Scope addition (2026-07-05) — make `study_power_model_compare.py` measure the out-of-the-box +method on the full campaign range.** The committed benchmark must reflect performance with **no +specialist tuning**: today the driver passes four accepted-by-findings behaviour settings that are +not `PowerModelMethod` class defaults, and the campaign grid stops at 3 months. +- **Promote the accepted behaviour defaults onto the class** so a bare `PowerModelMethod` + behaves like the benchmarked method: `TUNED_MODEL_PARAMS` (`min_child_samples=50`, F14) as the + `model_params` default and `availability_feature=False` (F13) directly; make + `CURATED_ERA5_EXCLUDE` defaultable by softening the exclusion to drop-if-present (it currently + raises on non-Open-Meteo frames, which is why F13 left it driver-level). `reference_stat_cols` + stays driver-level — it names a source-specific SCADA tag, i.e. data-schema description, not + tuning; document that distinction where the drivers configure it. After promotion the driver + should pass **only** schema description (columns, rated power, ERA5 frame, stat-col names). +- **Extend the scored campaign grid to 1/2/3/6/12 months in both modes** (toggle drops 9): relax + the alignment guard to check truth equality on *intersecting* cases only (still failing loudly + on any truth mismatch); **compute `naive_ratio` fresh** for campaign lengths the frozen + reference run does not cover (it is cheap — v0 stays reference-only and is simply absent at + 1–2 months, where comparing to naive suffices); regenerate the committed benchmark JSON on the + new grid. The F16 short-campaign regime evidence (`inspect_short_campaigns.py`) then becomes + part of every future A/B automatically — which Issue 15's adaptive rule needs. +- **Sequencing:** the three fixes interact (balance changes effective counts → floor; + floor changes coverage → re-level), so land and measure them incrementally in the + order re-level fix (a pure bug) → floor + imputation → balance, per the Issue 9 + protocol. Unit tests for the floor, the imputation, and the reweighting; A/B via + `study_power_model_compare.py` as usual. + +**Done when:** the sparse-extreme bins are no longer the "worse than benchmark" set +(fixed, or flagged-and-imputed by the floor); conditional |bias| no worse on +commonly-covered bins; coverage visible in the leaderboard; the decomposition — +measured plus pinned imputed bins — still energy-aggregates exactly to the headline. +Additionally (the 2026-07-05 scope addition): a bare `PowerModelMethod` given only data-schema +config behaves identically to the benchmarked configuration; `study_power_model_compare.py` +scores 1/2/3/6/12-month campaigns in both modes with naive computed fresh where the reference +lacks it; the benchmark JSON is regenerated on the new grid. + +*Shipped 2026-07-08 (findings F17): count floor (`_MIN_BIN_MATCHED_COUNT=50`) + physics imputation + +corrected pinned-imputed re-level — overall P50 bit-identical, conditional score −3.7 pp prepost / +−2.6 pp toggle, sparse-extreme "worse" bins eliminated (0 worse). Coverage kept method-internal (per-run +CSV), not surfaced to the leaderboard (user decision). Per-bin balance **deferred** — floor cleared the +done-when, so adopt-only-if-it-helps left it unbuilt (follow-up). Defaults promoted to the class; grid +1–12 months both modes with fresh naive; benchmark regenerated. Note: the frozen 30-June reference dir is +stale (its v0 needs regenerating on current code — see F17).* + +--- + +## Issue 15 — The headline estimator configures itself: regime-adaptive training window and estimator (WS2) + +**Goal:** the method must "just work" on any campaign with **no manual configuration** — +a user should not need to know when to flip `toggle_campaign_only`, pick a +`toggle_estimator` or tune a decay half-life. F16 measured why this matters: the best +configuration is regime-dependent, with a crossover near ~3 months. + +**The measured regime map (F16, `inspect_short_campaigns.py` + the benchmark grid):** +- Toggle training window: campaign-only wins at 1–2 months (the campaign is <5 % of the + all-data training set, so drift dominates; 1-month score 0.334 vs 0.895), all-data wins + at ≥3 months (shrinkage dominates; 3-month 0.294 vs 0.359). +- `double_ratio` (the model-normalised naive comparison): toggle placebo headline bias + ≈ 0 at every 3–12-month length, but *worse* than the default at 1–2 months — its + `rho_off` is currently measured over all off rows (two years of eras) while the ON + window is a sliver, so the cancellation subtracts the wrong era's miscalibration. +- Decay half-life: ~90 days is best-or-near-best at 1/2/3 months in both modes; ≥1 year + is the safe long-campaign default (548 d accepted in F16). + +**Scope (in preference order — a dominating single configuration beats selection logic)** +- **Era-local `rho_off` first.** Fix `double_ratio` itself: measure `rho_off` on + campaign-window off rows only (out-of-fold), predicted by the same all-data-trained + fold models. If that removes the 1–2-month failure while keeping the ≥3-month bias ≈ 0, + one estimator may dominate everywhere and become the toggle default outright — no + selection machinery needed. Test at 1, 2, 3, 6, 12 months in both framings. +- **Adaptive training window / half-life as fallback.** If no single configuration + dominates: derive a blended weight from the campaign's own observables + — campaign length, campaign-row count vs pre-campaign-row count — never from the + benchmark truth. Prefer a smooth blend (e.g. weight the campaign-only and all-data + estimates by a logistic in log(campaign rows)) over a hard threshold, so estimates + don't jump discontinuously at a boundary; hard-validate whatever rule is chosen at the + boundary lengths (2–4 months). +- **Guard against benchmark overfitting** (the Issue 12 framing rule): the rule's + parameters must be justified by the mechanism (drift-vs-shrinkage trade-off), tuned at + most coarsely, and confirmed on profiles/turbines not used to pick them. +- **Evaluation surface:** extend the scored campaign grid to include 1- and 2-month + campaigns for the adaptive-rule validation (either promote them into the committed + benchmark — needs fresh naive/v0 reference runs — or keep `inspect_short_campaigns.py` + as the reproducible driver for that regime, citing it in the findings entry). +- Remove the user-facing knobs this makes redundant (or demote them to expert overrides + with the automatic behaviour as the default). + +**Done when:** the out-of-the-box configuration is best-or-tied (within the materiality +band) at every campaign length 1–12 months in both modes on the placebo and standard +profiles; no scenario requires the user to choose a training window, estimator or +half-life; the findings entry records the mechanism-based justification for the rule (or +the dominance of the single configuration); benchmark regenerated. + +--- + +## Issue 16 — Prune the historic opt-ins: orient `power_model` to the winning configuration (power_model) + +**Goal:** the accumulated development knobs on `PowerModelMethod` are almost all opt-ins that +**lost their A/B and are off in the shipped default**. Remove the dead ones so the code presents +only the successful approach — smaller surface, less overfit temptation, easier to reason about. +This is code hygiene, not a methodology change: the default configuration is unchanged, so the +committed benchmark must come out **bit-identical** (the acceptance test). Also finishes the +Issue 15 knob cleanup. + +**The removal set (each off/absent in the default; findings verdict in brackets)** +- `calibrate_slope` [F14 — toggle overcorrects; OOF-vs-final shrinkage gap is structural], + `calibrate_residuals` [F15 — OOF residuals don't transfer to a refit model in level or shape]. +- `early_stopping` [F14 — neutral even in its target 3mo-toggle regime, +15% runtime], + `n_seed_ensemble` [F14 — spread is weather-sampling not seed noise, +75% runtime]. +- `toggle_estimator="double_ratio"` **and** the temporary `rho_off_scope` knob [F16/F19 — never + beats the counterfactual default on score; boxed between "nothing to correct" (all-data → + `rho_off`≈1) and "too noisy to correct" (campaign-only)]. Removing `double_ratio` makes + `toggle_estimator` a single-value knob → drop it entirely; the counterfactual is the sole + headline estimator. +- `time_features` (+ `latitude`, `longitude`) [F11 — all rejected: prepost drift import, calendar + proxy], `era5_derivations` (+ `hub_height_m`) [F10 — all rejected *as model features*]. + +**Keep (with rationale, so it isn't re-litigated)** +- The **injectable `model_factory` seam** (roadmap Phase-2 learners) *may* stay, but drop the dead + `hgb`/`linear` registry entries and the calibration/early-stop/seed plumbing in `fitting.py`, + keeping only the shared `time_block_folds`. Decide seam-stays-vs-goes on whether any driver needs + it; if nothing uses it, remove it too. +- **`era5_derived.py` and `time_features.py` stay as utilities** — CEM matching / the inspection + scripts use the derivations even though the *feature knobs* are removed. Only the model-feature + wiring and the method-surface knobs go. +- Settled defaults stay: `era5_exclude=CURATED_ERA5_EXCLUDE` (F13), `availability_feature=False` + (F13), `reference_stat_cols` (schema), `matching_vars`/`matching_bin_edges` (F6), the adaptive + time-decay default and its `time_decay_half_life_days` expert override (F20). + +**`toggle_campaign_only` — confirm then likely remove.** The adaptive half-life (F20) is a *soft* +campaign-only at short campaigns, so `toggle_campaign_only=True` is probably now redundant for +`power_model`. One cheap A/B (adaptive vs adaptive+campaign-only, toggle 1–3mo, both placebo and a +recovery profile) settles it; remove the knob if redundant, keep with fresh evidence if not. +(`naive_ratio` keeps its own campaign-only restriction — a different method, untouched.) + +**Scope** +- Delete the dead knobs, their config-validation branches, their `_config_params` entries, their + plumbing, and their unit tests; thin `fitting.py`; update any driver/inspection script that set a + removed knob (e.g. `inspect_short_campaigns.py`'s `double_ratio*` arms). +- Re-run `study_power_model_compare.py` on the default config and assert the benchmark is + bit-identical (no `--update-baseline` needed — nothing should change). + +**Done when:** the removed knobs are gone with their tests; `poe all-fast` green; a default-config +run of `study_power_model_compare.py` reproduces the committed benchmark bit-identically; a short +findings entry records the pruned set and the `toggle_campaign_only` verdict. + +--- + +## Issue 17 — Make the conditional estimator consistent with the headline (power_model) + +**Goal:** the per-bin conditional uplift and the overall headline currently use **different data and +weighting**, an asymmetry that grew up historically (F8/F15/F16) and is now partly obsolete. Reconcile +them — either unify the data/weighting path, or keep the asymmetry only where fresh evidence still +demands it and document why. + +**The current asymmetry (`method.py:_estimate_conditional`)** +| aspect | headline (overall) | conditional (per-bin) | +| --- | --- | --- | +| training data | **all** baseline | campaign-window rows **only** (F15) | +| recency weighting | **adaptive half-life** (F20) | **none** (F16) | +| estimator | single full-window fit | two-direction CEM-matched cross-prediction, re-levelled to the headline | + +The re-level (F8/F14) keeps the *aggregate* consistent; the *inputs* diverge. + +**Why this is an investigation, not a one-liner** +- **The missing half-life is downstream of the campaign-only matching.** The conditional already + excludes pre-campaign rows, so a decay weight has little to act on — adding the half-life is only + meaningful *if* the conditional also starts using pre-campaign data like the headline. +- **F15 localised the all-data damage to the *matching*, not the fits** (matching pre-campaign rows + against campaign on-rows reads reference/era drift as per-bin uplift). So naive unification + (feed it all-data + decay) is known to fail; genuine unification needs *weighted or restricted* + matching — design work. +- **But the narrow version is now unblocked:** Issue 14 deferred lifting the "weights are + headline-fit-only" restriction until the per-bin count floor existed ("the floor should make the + conditional path robust enough to lift that restriction"); the floor shipped in **F17**. So + re-trialling recency weighting on the conditional path is ready. + +**Scope (per the Issue 9 A/B protocol, one change at a time)** +- **Re-trial recency weighting on the conditional fits** now the F17 count floor is in place — the + specific F14/F16-flagged item. Watch the sparse extreme-TI/ws tail bins that destabilised before + the floor. +- **Trial unifying the training window**: let the conditional use all-data with the adaptive half-life + *and* an appropriately weighted/restricted CEM match, vs today's campaign-only match. The win + condition is the F15 drift bias staying cancelled while the extra data cuts per-bin spread. +- **Preserve the exact energy-aggregation identity** (measured + pinned-imputed bins aggregate to the + headline) throughout — it is the one consistency property already correct and must not regress. +- Evaluate on the conditional benchmark (`conditional_benchmark_comparison`) across profiles and + campaign lengths; gate on conditional |bias| no worse (ideally better) and overall P50 unchanged. + +**Done when:** either the conditional and headline share one data/weighting path (with the energy +identity intact and conditional |bias| no worse), **or** each surviving asymmetry is deliberately +documented with fresh post-floor evidence for why it must remain; benchmark regenerated if any default +changes; findings entry records the verdict. +Campaign curve plots for the winning approach are re-generated for user inspection. + +--- + +## Issue 18 — AEP uplift: long-term extrapolation design + harness scoring (WS1/WS4) + +**Goal:** turn the campaign conditional uplift into an **AEP uplift** — the long-term +annual energy benefit, PR #100's third metric — and score it in the harness against known +ground truth. The core idea: choose the condition axes from the nature of the upgrade +(wind speed only for most upgrades; add wind direction etc. for complex upgrades like +wake steering), estimate the **long-term MWh distribution** over those condition bins, +and take the energy-weighted sum of the conditional uplift: +`AEP uplift = Σ_b E_b^LT · (1+u_b) / Σ_b E_b^LT − 1`. + +**Scope** +- **Upgrade-nature-driven condition axes.** A config on the method/driver naming the + binning axes for extrapolation (default `("ws",)`; wake steering `("ws", "wd")`, …). + Generalise the Issue 8 conditional machinery to produce `u_b` on those axes (today it + emits ws and TI *marginals*; direction and joint axes need the same treatment). +- **Long-term energy distribution `E_b^LT`.** Two candidate sources, decided in design: + (a) long on-site SCADA history binned on measured conditions (simple, but needs years + of history and stationary sensors); (b) long-term ERA5 (10–20 y) driven through the + campaign-learned relation between ERA5 and the test turbine's conditions/energy — e.g. + the baseline counterfactual model evaluated over the long-term ERA5 record, or an + ERA5-cell occupancy map × per-cell mean energy from the campaign. (b) is preferred + (works for any site, matches the WS4 density-ratio framing). +- **The axis-consistency question (design decision).** Conditional uplift is binned on + test-measured ws/TI, but the long-term record is ERA5 — either learn the ERA5→test-axis + transfer on the campaign overlap, or run the AEP path end-to-end on the ERA5 axis + (matching, uplift bins and long-term weights all on the same treatment-invariant axis). + Sketch both in a short design note section before coding; the ERA5-axis route avoids a + post-treatment reporting axis entirely and is likely cleaner for AEP. +- **Coverage fallback.** Long-term bins with no measured `u_b` (or floored by Issue 14) + take Issue 14's imputed values (bfill-then-0 on the ws axis — the 0-at-rated prior + matters even more here, since long-term energy concentrates in high-ws bins that an + overall-uplift fallback would over-credit); report the coverage fraction (share of + long-term energy in bins with a measured uplift) as a headline diagnostic. +- **Harness truth + scoring.** The generator knows the true uplift function, so the true + AEP uplift per replicate is the injected profile applied over the **full multi-year + dataset** (not just the campaign window); score `estimate − truth` with the same + bias/spread/score metrics and campaign-length sweep. Include a naive extrapolation + baseline (`AEP uplift = campaign overall uplift`) — the bar to beat, which only a + condition-dependent profile can separate from the real thing. +- **A direction-dependent (wake-steering-shaped) profile** in the generator, so the + "bin by direction" path is actually exercised (listed in Issue 1, never implemented). + +**Done when:** the design decision (axis strategy) is recorded; the harness emits AEP +truth and scores; `power_model` produces an AEP estimate on at least the ws axis and +beats the naive extrapolation baseline on condition-dependent profiles. + +--- + +## Issue 19 — Uncertainty: campaign & AEP P50 uncertainty with harness coverage scoring (WS4) + +> **Partly delivered ahead of this issue, for `toggle_specialist` only (2026-07-15).** A +> toggle-first slice built the machinery this issue describes: the additive seam extension +> (`MethodOutput.sigma_overall` / `uncertainty_diagnostics`), the circular block bootstrap +> (`benchmarking/baselines/block_bootstrap.py`), harness calibration scoring +> (`benchmarking/harness/calibration.py`), and a 64-replicate coverage study +> (`study_toggle_specialist_uncertainty.py`). `toggle_specialist` now reports a **non-optional** +> 1σ on the headline and every power bin. **Read F28/F29 in `findings.md` before planning the +> rest** — they change three assumptions written below: +> +> - **"block length ≥ the residual autocorrelation scale, likely ~1 day" is wrong for a fast +> toggle.** The *paired* on/off residual decorrelates in ~1-3h (autocorr 0.003 at 48h), so a +> day-scale block captures nothing and biases σ **low** via circular-overlap collapse. There is +> no σ-vs-L plateau to read; block length must be picked by coverage against truth. +> - **The scored quantity should be 1σ coverage (~68.3%), not only P95.** A one-sided P95 check +> is a weak instrument: it tests one tail at 95%, so it needs ~4x the ensemble to resolve the +> same miscalibration, and it cannot distinguish a scale error from a shape error. The measured +> errors are **platykurtic** (kurtosis -0.35..-0.76), which means P95 ≠ P50 − 1.645σ even when σ +> is correctly scaled — so the normality shortcut below must be checked, not assumed. +> - **"Validate first on the placebo" buys little.** The three profiles' errors correlate +> 0.977-0.995, so the placebo is not an independent test; replicates are the only real evidence +> axis, and the ensemble must be sized on them (n=4 produced a phantom -0.7 pp bias that +> vanished at n=64). +> +> Still open here: the *cheap vs full* bootstrap comparison, `power_model` uncertainty, AEP σ, and +> P95 itself. + +**Goal:** start Phase 3 — report an uncertainty (σ / P95) on the campaign overall P50 and +the AEP P50, and verify it in the harness per PR #100: the campaign-uplift P95 should sit +below the true uplift ~95% of the time. (P50 accuracy work stays the priority; this issue +establishes the machinery and a first honest number, not the final uncertainty model.) + +**Scope** +- **Method-side estimator: circular block bootstrap** over time blocks of the baseline + and upgraded rows (block length ≥ the residual autocorrelation scale, likely ~1 day). + Two variants to compare on a small grid: *cheap* (fix the fitted model, resample the + (actual, counterfactual) pairs → distribution of the energy ratio) and *full* (refit the + model per resample — captures model-fit noise, ~50–100× the cost). Emit σ and empirical + quantiles; P95 = P50 − 1.645σ under normality or the empirical quantile, whichever the + data supports. +- **Seam extension (additive).** Optional uncertainty fields on `MethodOutput` + (e.g. `sigma_overall`, `p95_overall`, per-bin σ) — `None` for methods that don't emit + them, no breaking change to existing methods. +- **Harness scoring.** Coverage = fraction of (replicate × campaign) cases with + `truth ≥ P95` (target 0.95), plus a calibration read at 2–3 more quantiles using the + replicate ensemble, plus mean interval width — so a method cannot win coverage by + inflating σ. Validate first on the placebo (truth 0), where miscalibration is most + visible. +- **Why not quantile regression for P95:** per-row predictive quantiles describe + single-timestamp scatter and do not aggregate into a quantile of the campaign *ratio* + without independence assumptions the autocorrelated 10-min data violates — hence the + block bootstrap. Quantile models stay on the WS4 list for conformal OOD filtering and + diagnostics. +- **AEP P95.** Combine the campaign sampling uncertainty (bootstrap above) with the + long-term-distribution uncertainty (resample ERA5 *years* to get the inter-annual + variability of `E_b^LT`). Document explicitly what is in and out of scope of the + reported number (e.g. model-form error and sensor drift are out, for now). +- Cross-check the block bootstrap design against v0's existing bootstrap (`wind_up` + uncertainty machinery) — same idea, method-appropriate implementation; no v0 import. + +**Done when:** `power_model` emits σ/P95 for campaign and AEP uplift; the harness reports +coverage and interval width per profile × campaign length; observed coverage is within an +agreed tolerance of nominal on the placebo and the standard profiles. + +--- + +## Not in the first wave (tracked for later phases) + +- Uncertainty / P95 model beyond Issue 19's first cut: conformal OOD filtering, + density-ratio long-term weighting in place of hard CEM subsampling (WS4, Phase 3). + Density-ratio weighting also reclaims the short-campaign matched-set size the CEM + subsample throws away (F7 follow-up). +- Further candidate methods: DSWE / `funGP`, Astolfi multivariate-linear (WS2, + Phase 2). +- Conditional/heterogeneous uplift reporting & SHAP story beyond ws/TI/direction + bins (G3, Phase 2). +- Report content generation and I/O / step-independence maturation (WS5, Phase 4). +- **Method-controlled baseline horizon / recency weighting (R-learner).** Today the R-learner + pools *all* pre-upgrade baseline it is given (e.g. 24 months) into the nuisance fits with equal + weight, and uses no timestamp features, so it cannot down-weight stale data. Since the goal is + usually the upgrade's *future* benefit, the most recent baseline is the most representative of + the farm's future state, and far-past data (turbine ageing, sensor drift, soiling, controller + changes) is less so. Add a method-owned horizon control — a `max_baseline_months` cap and/or an + exponential recency weight on the fit (via LightGBM `sample_weight`) — so the method, not the + harness, decides how far back to trust. Keep it compatible with the no-timestamp-feature rule + (weighting/selection by recency is fine; calendar features are not) and the block bootstrap. + Pairs naturally with the long-term ERA5 extrapolation work (representativeness weighting, WS4). +- ~~**Derived ERA5 atmospheric features.**~~ Promoted into **Issue 9** (air density, shear + exponent, veer, stability indicators as a shared ERA5-derivation utility). + +--- + +## Future work + +- **Uncertainty is validated only on 10-minute data.** The `toggle_specialist` block-bootstrap + uncertainty (F28-F33) has been calibrated exclusively against Hill of Towie 10-minute SCADA — the + only timebase available to test against. Two pieces carry timebase assumptions that a coarser or + finer feed would disturb and that nothing currently re-checks: + - the **6-hour default block length** (F28) is tuned to the ~1-3h autocorrelation of the 10-minute + paired residual and the 40-minute toggle period; a different sampling rate or toggle cadence + shifts both; + - the **record-count blend range** (`_BLEND_LO_RECORDS=3`, `_BLEND_HI_RECORDS=7`, F33) is in units + of records, so its wall-clock meaning scales inversely with the timebase. + When non-10-minute data becomes available, re-run the calibration sweep on it and confirm (or + re-derive) these constants; consider expressing block length and the blend range in time units + rather than record counts so they travel across timebases by construction. diff --git a/docs/v1/references.md b/docs/v1/references.md new file mode 100644 index 00000000..ca7bd8f2 --- /dev/null +++ b/docs/v1/references.md @@ -0,0 +1,95 @@ +# wind-up v1 — related tools & prior art + +A living list of related open-source tools and key methodology references worth +deeper investigation as v1 progresses. Complements the reading list already in +`wind-up-ml-uplift-design-note.md` (causal-ML / treatment-effect literature). + +## Related open-source tools + +### FLASC — FLORIS-based Analysis for SCADA data (NREL) + + +SCADA filtering, analysis, model validation, and field-experiment design/monitoring, +integrated with the FLORIS wake model. Relevant to v1: +- **Synthetic / artificial data** — has `examples_artificial_data`; this is the + precedent (P. Fleming) for generating SCADA-like data with a known answer. + **Investigate for WS1** (synthetic upgrade-dataset generator): how they construct + artificial data and inject known effects. +- **Energy-ratio methodology** for quantifying wake effects in both synthetic and + historical data — a comparison point for uplift/energy-ratio metrics. +- Time-synchronization and filtering utilities for multi-turbine SCADA. + +### OpenOA — Operational Assessment (NREL) + + +Wind-plant operational assessment from SCADA + met + reanalysis. Relevant to v1: +- **`MonteCarloAEP`** — long-term AEP from 1–3 yr records with reanalysis (ERA5 / + MERRA-2) and **Monte-Carlo uncertainty**. Comparison/inspiration for **WS4** + (uncertainty) and long-term extrapolation (G4 step 3). +- **`PlantData` schema** — a standardized, validated data structure integrating + SCADA / met towers / revenue meters / reanalysis. **Prior art for WS3** (the + assessment data contract) — worth studying before fixing our schema. +- Utilities: power-curve fitting, outlier/range filtering, imputation, + met processing, plotting. +- Other analyses (`TurbineLongTermGrossEnergy`, `ElectricalLosses`, `WakeLosses`, + `EYAGapAnalysis`, `StaticYawMisalignment`) — context, not directly uplift. + +### DSWE — Data Science for Wind Energy (Y. Ding et al.) +R and Python. Implements the three-step `funGP` performance-comparison pipeline +(covariate matching → power model → functional-GP comparison with confidence bands), +explicitly framed as treatment-effect estimation. A natural **WS2 candidate / +cross-check**. See design-note §5. + +## Key methodology references + +### Kanev, *AWC validation methodology*, TNO 2020 R11300 (Aug 2020) +The reference for the **multi-dimensional binning** candidate (WS2). Directly +relevant beyond binning: +- **Multi-dimensional binning** — bin by wind speed *and* wind direction; keep ws + bins fixed at 1 m/s and **adaptively widen the wind-direction bin** until the + normalized standard error of farm power per bin drops below a target. Enables a + power curve *per direction sector* (a route to G3 conditional uplift). +- **Toggle-period study (§4.8)** — quantifies how toggle/campaign length affects + the mean power ratio and its 95% CI. **Directly relevant to G2** (shorter + campaigns) and to designing the WS1 short-campaign sweep. Notable cautionary + result: a **12-hour toggle period** badly biases results because each data set + then samples only one half of the **diurnal cycle** (different atmospheric + stability) — reinforces G3 (day/night conditioning) and is a pitfall the + synthetic-data harness should be able to reproduce. +- **Consensus wind speed/direction** from farm-wide nacelle anemometry — a + treatment-invariant reference-construction idea relevant to WS3 features. +- **Filtering rules** for unavailability / curtailment / power-boosting (exclude + affected turbines and those in their wakes, rather than dropping whole records). +- **Uncertainty quantification (§4.9)** — power-ratio CIs; context for WS4. + +### Astolfi et al. — multivariate-linear before/after method +- Astolfi, Castellani & Terzi (2018), *Wind Turbine Power Curve Upgrades*, + Energies 11(5):1300. +- Astolfi, Castellani, Fravolini, Cascianelli & Terzi (2018), *Computing the real + impact of wind turbine power curve upgrades: a SCADA-based multivariate linear + method and a vortex generator test case* (preprints.org 2018060082). + +The source of the **WS2 "Astolfi multivariate-linear" candidate**. Key ideas: +- Don't compare raw before/after energy under non-stationary conditions; instead + compare post-upgrade production to a **data-driven model of pre-upgrade + production under the same conditions** (a before/after residual method) — the + same logic as wind-up's pre/post approach. +- **Multivariate linear** power model. Practical tip (design note §5): feeding + **min/max/std of the inputs, not just the means**, cut their error metrics by + roughly a third — directly relevant to feature engineering in **WS1/WS3**. +- Reported upgrade magnitudes (~1.3% yaw optimisation, ~2.5% control re-powering) + are the same order as Hill of Towie (~0.7–1.7%) — useful sanity range. + +### Ding, Barber & Hammer (2022) — funGP performance comparison +*Data-Driven wind turbine performance assessment and quantification using SCADA +data and field measurements*, Frontiers in Energy Research 10:1050342. + +The methodology paper behind the **DSWE / funGP** tool above: a three-step +pipeline — covariate matching → data-driven power model → **functional-GP +(`funGP`) comparison with confidence bands** — explicitly framed as +treatment-effect estimation. A **WS2 candidate / cross-check**. + +### Further reading +See `wind-up-ml-uplift-design-note.md` §5 for the causal-ML reading list (R-learner, +DML, metalearners; Ding/Astolfi wind-specific work; conformal prediction; block +bootstrap) and §7 for the proposed dependencies. diff --git a/docs/v1/roadmap.md b/docs/v1/roadmap.md new file mode 100644 index 00000000..bc92ea06 --- /dev/null +++ b/docs/v1/roadmap.md @@ -0,0 +1,88 @@ +# wind-up v1 — roadmap + +This file decomposes v1 into workstreams (epics) and lays out the phasing. See +[goals.md](goals.md) for the goals these serve, and [issues.md](issues.md) for the +concrete first issues. + +## Workstreams (epics) + +### WS1 — Evaluation harness *(public; critical path)* +The objective yardstick for everything else. Two parts: +- **Synthetic upgrade-dataset generator** — take real SCADA from stable, + no-upgrade periods (Hill of Towie and other open wind farms) and inject a + **known** uplift to create realistic datasets with a ground-truth answer. + Support a range of injected uplift *profiles*: constant Cp change, + wind-speed-dependent Cp change (region-2-only, tailing to 0 at rated — the AeroUp + shape), condition-dependent change (e.g. wake-direction-dependent — the + wake-steering shape; or stability-dependent, as TuneUp may be), rated-power + change, etc. +- **P50 scoring + short-campaign robustness sweep** — given a method's estimate + and the known injected truth, compute **accuracy** (bias) and **precision** + (spread) metrics; sweep campaign length to measure how each degrades with less + data (serves G2). + +Designed to stand alone so external collaborators can compete on it (WeDoWind / +Kaggle-style), mirroring the Hill of Towie power-prediction challenge. + +### WS2 — Methods +The competing uplift estimators, all judged on WS1. +- **v0 binned power-curve method as the baseline** — the bar to beat. +- **Candidate: cross-fit R-learner (P50)** — treatment-effect framing from the + design note; treatment-invariant reference-only features (never the test + turbine's own SCADA wind speed); LightGBM nuisances. +- **Later candidates** — DSWE / `funGP` as a cross-check, Astolfi + multivariate-linear residual method, others. +- Multi-dimensional binning (i.e. the AWC validation methodology by S. Kanev, + TNO 2020 R11300, Aug 2020) — wind speed × wind direction binning with adaptive + bin sizing. See [references.md](references.md). + +### WS3 — Minimal foundations *(only what unblocks WS2)* +- **Assessment data contract** — a standardized per test-reference, pre/post + conditioned dataset (with treatment-invariant features) that ANY method + consumes. This is the interface methods plug into. +- **Pluggable method selector** — a thin `assessment_method` config field that + picks the estimator, reusing existing inputs / test-ref pairing / result + objects. Kept deliberately minimal; not a big refactor (see goals.md scope + guards). + +### WS4 — Uncertainty / P95 *(deferred)* +Begins only after a P50 winner is identified, because the uncertainty model +depends on the chosen point method. From the design note: block bootstrap +(autocorrelation-aware), conformalized quantile regression for OOD filtering, and +density-ratio weighting for long-term (ERA5) extrapolation under covariate shift. + +### WS5 — Reporting & I/O maturation *(parallel-later)* +- **Report content generation (G6)** — auto-produce the report pieces currently + made by hand: executive-summary numbers, per-turbine results table, combined and + by-pair uplift charts, farm-layout figure with test/reference highlighting, + exclusion-period table, reference-suitability table. +- **I/O & config maturation (G5)** — cleaner inputs/outputs and configuration. +- **Pipeline step independence (G4)** — run preprocess / measure / long-term / + aggregate as standalone steps. + +## Phasing + +| Phase | Focus | Workstreams | +|-------|-------|-------------| +| **1 — now** | Harness + baseline + first new method, on **P50 only** | WS1, WS2 (baseline + R-learner), WS3 (minimal contract) | +| **2** | More methods; conditional/heterogeneous uplift (G3); short-campaign study (G2) | WS2, WS1 | +| **3** | Uncertainty / P95 model | WS4 | +| **4 (overlap)** | Reporting (G6), I/O maturation (G5), step independence (G4) | WS5 | + +## Dependency notes + +- WS1 (harness) gates WS2 — no objective comparison without it. +- WS3 (data contract) is the seam between WS1 and WS2: methods consume the + contract; the harness feeds synthetic data through it. +- WS4 strictly follows a P50 winner from Phase 1–2. +- WS5 is independent and can proceed whenever there is appetite, since it builds + on the existing v0 outputs. + +## Management + +- All v1 PRs target the `v1` branch. +- First issues are drafted in [issues.md](issues.md) and created on GitHub + (`resgroup/wind-up`) once wording is settled — suggested labels `v1`, + `WS1-harness`, `WS2-methods`, `WS3-foundations`, and a `v1` milestone. +- Implementation of each issue should go through the normal design → plan → build + flow in its own session. diff --git a/pyproject.toml b/pyproject.toml index 2c13e4b4..a87e41cf 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -46,6 +46,28 @@ requires = ["setuptools>=61.0"] build-backend = "setuptools.build_meta" [project.optional-dependencies] +era5 = [ + 'openmeteo-requests', + 'requests-cache', + 'retry-requests', +] +ml = [ + 'lightgbm', + 'scikit-learn', +] +examples = [ + 'jupyterlab', + 'notebook', + 'ipywidgets', + 'requests', + 'ephem', + 'flaml[automl]', +] + +# PEP 735 dependency groups. uv installs the `dev` group by default on `uv sync` (use +# `--no-dev` to skip it), so contributor tooling is present without naming an extra; unlike +# `[project.optional-dependencies]`, groups are not published in the wheel metadata. +[dependency-groups] dev = [ 'pytest', 'coverage', @@ -59,19 +81,21 @@ dev = [ 'mypy<1.19', 'requests', 'pytest-env', -] -examples = [ - 'jupyterlab', - 'notebook', - 'ipywidgets', - 'requests', - 'ephem', - 'flaml[automl]', + 'openmeteo-requests', + 'requests-cache', + 'retry-requests', + 'lightgbm', + 'scikit-learn', ] [tool.setuptools.packages.find] where = ["."] -include = ["wind_up*"] +# `benchmarking*` is included temporarily so downstream consumers can import the benchmarking +# methods from an editable/path install. This DOES add `benchmarking` to any wheel/sdist built +# from this repo; it is acceptable only because v1 is unreleased (nothing is published yet). It +# MUST be removed before v1 is released so the package is not shipped. +# TODO clean up the benchmarking package and remove this once v1 is ready for release. +include = ["wind_up*", "benchmarking*"] [tool.uv] # Dev/CI-only constraint (not part of published metadata, so consumers are unaffected). @@ -133,12 +157,17 @@ module = [ "pandas", "pandas.testing", "ruptures.*", + "scipy.interpolate", + "scipy.ndimage", "scipy.stats", "seaborn", "utm", "ephem", "flaml", "sklearn.*", + "openmeteo_requests", + "requests_cache", + "retry_requests", ] disable_error_code = ["name-defined"] ignore_missing_imports = true diff --git a/tests/benchmarking/__init__.py b/tests/benchmarking/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/benchmarking/baselines/__init__.py b/tests/benchmarking/baselines/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/benchmarking/baselines/test_block_bootstrap.py b/tests/benchmarking/baselines/test_block_bootstrap.py new file mode 100644 index 00000000..24ce0640 --- /dev/null +++ b/tests/benchmarking/baselines/test_block_bootstrap.py @@ -0,0 +1,382 @@ +"""Tests for the circular block bootstrap behind the toggle specialist's uncertainty. + +The bootstrap is pure numerics on paired ``(test, reference)`` sums, so these build the timeline +directly rather than going through SCADA. They cover the recovered scale on data with a known +sampling error, the block-length and campaign-length responses, the pairing property that makes the +whole design work, reproducibility, and the degenerate paths. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.block_bootstrap import bootstrap_ratio_uplift, relative_scatter + +_TIMEBASE = pd.Timedelta(minutes=10) + + +def _timeline(n: int) -> pd.DatetimeIndex: + return pd.date_range("2020-01-01", periods=n, freq=_TIMEBASE, tz="UTC") + + +def _alternating(n: int, *, block: int = 2) -> tuple[np.ndarray, np.ndarray]: + """On/off masks alternating every ``block`` records, as a fast toggle produces.""" + cycle = (np.arange(n) // block) % 2 + return cycle == 1, cycle == 0 + + +def _case( + n: int = 2016, + *, + uplift: float = 0.05, + noise: float = 0.05, + seed: int = 0, + ref_level: float = 500.0, +) -> dict: + """A toggle timeline whose test power is ``k * ref`` (times ``1+uplift`` when on) plus noise. + + Reference power is a slow sinusoid (a weather-like signal both segments share) so the on/off + pairing has something real to cancel; the noise is per-record and independent, which is the + variability the bootstrap should recover. + """ + rng = np.random.default_rng(seed) + times = _timeline(n) + on, off = _alternating(n) + ref = ref_level * (1.2 + np.sin(np.arange(n) / 144.0)) + test = 0.8 * ref * np.where(on, 1.0 + uplift, 1.0) * (1.0 + noise * rng.standard_normal(n)) + return { + "times": times, + "test_power": test, + "ref_total": ref, + "upgraded": on, + "baseline": off, + "cell_membership": {"overall": np.ones(n, dtype=bool)}, + "campaign_start": times[0], + "campaign_end": times[-1], + "timebase": _TIMEBASE, + } + + +def _run(case: dict, *, block_hours: float = 48.0, n_resamples: int = 400, seed: int = 0, **kw: object): # noqa: ANN202 + return bootstrap_ratio_uplift(**case, block_hours=block_hours, n_resamples=n_resamples, seed=seed, **kw) + + +class TestSigma: + def test_sigma_is_finite_and_positive(self) -> None: + cell = _run(_case()).cells["overall"] + assert np.isfinite(cell.sigma) + assert cell.sigma > 0 + assert cell.frac_resamples_finite == 1.0 + + def test_sigma_matches_the_actual_scatter_of_the_estimator(self) -> None: + """The point of a bootstrap: its sigma should equal the estimator's real sampling spread. + + Measured by re-drawing the noise many times and taking the spread of the resulting point + estimates, which is the quantity a single run's bootstrap is trying to infer from one draw. + """ + estimates = [] + for seed in range(60): + case = _case(seed=seed) + on, off = case["upgraded"], case["baseline"] + test, ref = case["test_power"], case["ref_total"] + rho_up = test[on].sum() / ref[on].sum() + rho_base = test[off].sum() / ref[off].sum() + estimates.append(rho_up / rho_base - 1.0) + actual_spread = float(np.std(estimates, ddof=1)) + + sigma = _run(_case(seed=0), n_resamples=2000).cells["overall"].sigma + assert sigma == pytest.approx(actual_spread, rel=0.35) + + def test_sigma_shrinks_with_campaign_length(self) -> None: + short = _run(_case(n=1008)).cells["overall"].sigma + long = _run(_case(n=8064)).cells["overall"].sigma + assert long < short + # ~8x the records should cut sigma towards 1/sqrt(8); allow a wide band since the block + # bootstrap also has fewer blocks to work with at the short end. + assert 1.5 < short / long < 4.5 + + def test_robust_sigma_agrees_with_sigma_for_a_well_populated_cell(self) -> None: + """They diverge only when the resample distribution is heavy-tailed; here it should not be.""" + cell = _run(_case(), n_resamples=2000).cells["overall"] + assert cell.sigma_robust == pytest.approx(cell.sigma, rel=0.2) + + +class TestPairing: + def test_a_block_carries_its_on_and_off_rows_together(self) -> None: + """The design's load-bearing property. + + A slow shared signal in the reference cancels between on and off *within* a block. If blocks + did not carry both segments, that signal would leak into sigma and inflate it. Making the + shared signal far larger must therefore barely move sigma. + """ + calm = _run(_case(ref_level=500.0)).cells["overall"].sigma + # Same noise, same uplift, a much stronger shared weather signal on both segments. + wild = _run(_case(ref_level=5000.0)).cells["overall"].sigma + assert wild == pytest.approx(calm, rel=0.25) + + +class TestBlockLength: + def test_block_length_does_not_change_the_point_estimate_inputs(self) -> None: + """Sigma varies with block length; nothing else the bootstrap touches does.""" + sigmas = {bl: _run(_case(), block_hours=bl).cells["overall"].sigma for bl in (6.0, 24.0, 48.0)} + assert all(np.isfinite(s) and s > 0 for s in sigmas.values()) + + def test_n_blocks_is_the_campaign_divided_by_the_block(self) -> None: + # 2016 records at 10min = 14 days; 48h blocks -> 7. + assert _run(_case(n=2016), block_hours=48.0).n_blocks == 7 + assert _run(_case(n=2016), block_hours=24.0).n_blocks == 14 + + def test_a_block_longer_than_the_campaign_falls_back_rather_than_claiming_certainty(self) -> None: + """One block spans the campaign, so no resample can vary: the bootstrap has nothing to say. + + It used to return ~1e-15 — float residue, which a reader sees as "+/-0.0 pp" and reads as + certainty. Now the bootstrap reports NaN and the fallback carries the cell. + """ + result = _run(_case(n=1008), block_hours=1000.0) + assert result.n_blocks == 1 + cell = result.cells["overall"] + assert np.isnan(cell.sigma_bootstrap) + assert np.isfinite(cell.sigma) + assert cell.sigma == cell.sigma_fallback + + +class TestReproducibility: + def test_same_seed_gives_the_same_sigma(self) -> None: + a = _run(_case(), seed=7).cells["overall"].sigma + b = _run(_case(), seed=7).cells["overall"].sigma + assert a == b + + def test_different_seeds_agree_to_monte_carlo_noise(self) -> None: + a = _run(_case(), seed=1, n_resamples=2000).cells["overall"].sigma + b = _run(_case(), seed=2, n_resamples=2000).cells["overall"].sigma + assert a == pytest.approx(b, rel=0.1) + + +class TestCells: + def test_each_cell_is_bootstrapped_independently(self) -> None: + case = _case() + n = len(case["times"]) + first_half = np.arange(n) < n // 2 + case["cell_membership"] = { + "overall": np.ones(n, dtype=bool), + "first": first_half, + "second": ~first_half, + } + cells = _run(case).cells + assert set(cells) == {"overall", "first", "second"} + # A half-sized cell has fewer records, so a wider sigma than the whole. + assert cells["first"].sigma > cells["overall"].sigma + + def test_a_cell_with_no_baseline_rows_reports_nan_and_flags_it(self) -> None: + """No off rows means no ``rho_base``, so no uplift and nothing to bootstrap.""" + case = _case() + n = len(case["times"]) + case["cell_membership"] = {"on_only": case["upgraded"].copy()} + cell = _run(case).cells["on_only"] + assert np.isnan(cell.sigma) + assert np.isnan(cell.sigma_robust) + assert cell.frac_resamples_finite == 0.0 + assert n > 0 # guard against an accidentally empty case + + def test_an_empty_cell_reports_nan(self) -> None: + case = _case() + case["cell_membership"] = {"empty": np.zeros(len(case["times"]), dtype=bool)} + cell = _run(case).cells["empty"] + assert np.isnan(cell.sigma) + assert cell.frac_resamples_finite == 0.0 + + +class TestDegenerate: + def test_no_records_gives_nan_cells(self) -> None: + result = bootstrap_ratio_uplift( + times=_timeline(0), + test_power=np.array([]), + ref_total=np.array([]), + upgraded=np.array([], dtype=bool), + baseline=np.array([], dtype=bool), + cell_membership={"overall": np.array([], dtype=bool)}, + campaign_start=pd.Timestamp("2020-01-01", tz="UTC"), + campaign_end=pd.Timestamp("2020-01-08", tz="UTC"), + timebase=_TIMEBASE, + block_hours=48.0, + n_resamples=100, + seed=0, + ) + assert result.n_blocks == 0 + assert np.isnan(result.cells["overall"].sigma) + + def test_too_few_resamples_for_a_spread_gives_nan(self) -> None: + assert np.isnan(_run(_case(), n_resamples=1).cells["overall"].sigma) + + def test_unsorted_input_is_sorted_rather_than_trusted(self) -> None: + """Blocks are contiguous in time, so the record order must not change the answer.""" + case = _case() + ordered = _run(case, n_resamples=800).cells["overall"].sigma + + rng = np.random.default_rng(0) + shuffle = rng.permutation(len(case["times"])) + shuffled = dict(case) + shuffled["times"] = case["times"][shuffle] + for key in ("test_power", "ref_total", "upgraded", "baseline"): + shuffled[key] = case[key][shuffle] + shuffled["cell_membership"] = {k: v[shuffle] for k, v in case["cell_membership"].items()} + + assert _run(shuffled, n_resamples=800).cells["overall"].sigma == pytest.approx(ordered) + + +class TestTooFewRecordsToFallBackOn: + """A cell too sparse to bootstrap must still get an honest, wide sigma — not 0, and not NaN (F33). + + Resampling draws whole *blocks*, so if a cell's records sit in one block, every resample scales + numerator and denominator together (rho = k*test / k*ref) and the ratio never moves: the bootstrap + returns near-total confidence in a number built from one or two points. Measured on real + campaigns: coverage 0.158 at one record per side and 0.237 at two (target 0.683), with one cell + reporting sigma exactly 0 while being 14 pp wrong. The fallback is what covers that regime. + """ + + def _sparse_case(self, n_per_side: int) -> dict: + """A case whose 'sparse' cell holds exactly ``n_per_side`` records on each side.""" + case = _case(n=2016) + n = len(case["times"]) + sparse = np.zeros(n, dtype=bool) + sparse[np.flatnonzero(case["upgraded"])[:n_per_side]] = True + sparse[np.flatnonzero(case["baseline"])[:n_per_side]] = True + case["cell_membership"] = {"overall": np.ones(n, dtype=bool), "sparse": sparse} + return case + + def test_a_one_record_cell_gets_a_finite_sigma_from_the_fallback(self) -> None: + """The headline fix: 0 or NaN would both be worse answers than a wide number. + + The bootstrap still *reports* its collapsed value rather than hiding it behind NaN — that is + an honest diagnostic, and ``max`` is what stops it reaching the caller. + """ + cells = _run(self._sparse_case(1)).cells + sparse = cells["sparse"] + assert sparse.sigma_bootstrap < sparse.sigma_fallback, "the bootstrap collapses on one record" + assert np.isfinite(sparse.sigma) + assert sparse.sigma == sparse.sigma_fallback + + def test_the_sparse_sigma_is_much_wider_than_the_well_populated_one(self) -> None: + cells = _run(self._sparse_case(1)).cells + assert cells["sparse"].sigma > 10 * cells["overall"].sigma + + def test_the_fallback_narrows_as_records_are_added(self) -> None: + """The fallback scales as sqrt(1/n_on + 1/n_off), so more data must mean less of it. + + Asserted on the fallback component, not the reported sigma: the report switches to the + bootstrap once the cell is populated enough (the floor), so it is not monotone by design. + """ + fallbacks = [_run(self._sparse_case(n)).cells["sparse"].sigma_fallback for n in (1, 2, 4, 8)] + assert fallbacks == sorted(fallbacks, reverse=True) + + def test_a_well_populated_cell_reports_its_bootstrap_alone(self) -> None: + """The load-bearing guarantee of the selection rule: no fallback contamination of the covered + regime (F32/F33). It reports the bootstrap *even when the fallback is larger* — a ``max`` would + inflate it here, over-covering, and a max tuned on this farm could over-inflate on another. + """ + cell = _run(self._sparse_case(1)).cells["overall"] + assert np.isfinite(cell.sigma_bootstrap) + assert cell.sigma == cell.sigma_bootstrap + + def test_both_components_are_always_reported(self) -> None: + """So a blend rule can be re-judged from a saved sweep without re-running it.""" + for cell in _run(self._sparse_case(1)).cells.values(): + assert np.isfinite(cell.sigma_fallback) + + def test_the_report_ramps_between_the_two_across_the_transition(self) -> None: + """A gradual blend, not a cliff: a mid-transition cell lies strictly between the components.""" + cell = _run(self._sparse_case(5)).cells["sparse"] # min 5 -> weight 0.5 + lo, hi = sorted((cell.sigma_bootstrap, cell.sigma_fallback)) + assert lo < cell.sigma < hi + + def test_the_reported_sigma_moves_monotonically_from_fallback_to_bootstrap(self) -> None: + """As records grow, the report leaves the fallback and reaches the bootstrap, no jump.""" + by_n = {n: _run(self._sparse_case(n)).cells["sparse"] for n in (3, 4, 5, 6, 7)} + assert by_n[3].sigma == pytest.approx(by_n[3].sigma_fallback), "pure fallback at the low end" + assert by_n[7].sigma == pytest.approx(by_n[7].sigma_bootstrap), "pure bootstrap by the high end" + sigmas = [by_n[n].sigma for n in (3, 4, 5, 6, 7)] + assert sigmas == sorted(sigmas, reverse=True), "narrows monotonically, no cliff" + + def test_a_fully_populated_cell_is_pure_bootstrap(self) -> None: + """The load-bearing guarantee: no fallback contamination of the covered regime (F32).""" + cell = _run(self._sparse_case(1)).cells["overall"] + assert cell.sigma == cell.sigma_bootstrap + + def test_the_finite_fraction_is_still_reported_so_the_reason_is_visible(self) -> None: + cell = _run(self._sparse_case(1)).cells["sparse"] + assert np.isfinite(cell.frac_resamples_finite) + + +class TestRelativeScatter: + def test_it_recovers_a_known_per_record_noise_level(self) -> None: + """The fallback is only as good as this: it is the campaign's own measured scatter.""" + case = _case(noise=0.05) + s_rel = relative_scatter( + case["test_power"], case["ref_total"], upgraded=case["upgraded"], baseline=case["baseline"] + ) + assert s_rel == pytest.approx(0.05, rel=0.15) + + def test_it_tracks_the_noise(self) -> None: + cases = [_case(noise=x) for x in (0.02, 0.05, 0.10)] + scatters = [ + relative_scatter(c["test_power"], c["ref_total"], upgraded=c["upgraded"], baseline=c["baseline"]) + for c in cases + ] + assert scatters == sorted(scatters) + + def test_it_survives_reference_power_near_zero(self) -> None: + """A per-record ratio would explode near cut-in; the ratio-of-sums form does not.""" + case = _case() + case["ref_total"][:100] = 1e-9 # a near-cut-in stretch + s_rel = relative_scatter( + case["test_power"], case["ref_total"], upgraded=case["upgraded"], baseline=case["baseline"] + ) + assert np.isfinite(s_rel) + + +class TestPerfectDataMayReportZero: + """Nothing may preclude a zero sigma by construction. + + With noiseless, perfectly-matched data the uplift really is determined, so 0 is the correct + answer and a model that cannot express it is mis-specified. There is no irreducible floor to + justify one either: F31 tested exactly that hypothesis over campaigns up to a year and found + sigma kept shrinking and kept tracking the error down to 0.135 pp. + + An earlier version NaN-ed a zero spread to trap the 1-record artefact. That punished this + legitimate case to catch that one; the fallback traps the artefact without the collateral. + """ + + def _noiseless(self) -> dict: + case = _case(noise=0.0) + # exactly k * ref, times (1 + uplift) when on: no scatter at all + case["test_power"] = 0.8 * case["ref_total"] * np.where(case["upgraded"], 1.03, 1.0) + return case + + def test_both_components_vanish_on_noiseless_data(self) -> None: + cell = _run(self._noiseless()).cells["overall"] + assert cell.sigma_bootstrap == pytest.approx(0.0, abs=1e-9) + assert cell.sigma_fallback == pytest.approx(0.0, abs=1e-9) + + def test_the_reported_sigma_reaches_zero(self) -> None: + cell = _run(self._noiseless()).cells["overall"] + assert cell.sigma == pytest.approx(0.0, abs=1e-9) + assert not np.isnan(cell.sigma), "0 is the right answer here, not 'cannot estimate'" + + def test_the_scatter_measure_itself_vanishes(self) -> None: + """The fallback is proportional to s_rel, so it can only reach 0 if this does.""" + case = self._noiseless() + s_rel = relative_scatter( + case["test_power"], case["ref_total"], upgraded=case["upgraded"], baseline=case["baseline"] + ) + assert s_rel == pytest.approx(0.0, abs=1e-9) + + def test_sigma_scales_down_smoothly_as_noise_falls(self) -> None: + """Approaching zero continuously, not hitting a floor.""" + sigmas = [] + for noise in (0.10, 0.05, 0.02, 0.005): + case = _case(noise=noise) + sigmas.append(_run(case).cells["overall"].sigma) + assert sigmas == sorted(sigmas, reverse=True) + assert sigmas[-1] < sigmas[0] / 10, "no floor is arresting the descent" diff --git a/tests/benchmarking/baselines/test_era5_derived.py b/tests/benchmarking/baselines/test_era5_derived.py new file mode 100644 index 00000000..8aee9b8c --- /dev/null +++ b/tests/benchmarking/baselines/test_era5_derived.py @@ -0,0 +1,140 @@ +"""Known-input tests for the shared ERA5 derivation utility (Issue 9).""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.era5_derived import ( + ERA5_DERIVATIONS, + air_density, + era5_derived_frame, + gust_margin, + gust_ratio, + hub_height_wind_speed, + shear_exponent, + vertical_veer, +) + +_HOT_HUB_HEIGHT_M = 59.0 + + +def _series(values: list[float]) -> pd.Series: + return pd.Series(values, index=pd.date_range("2020-01-01", periods=len(values), freq="10min", tz="UTC")) + + +class TestShearExponent: + def test_known_value(self) -> None: + # ws doubles from 10 m to 100 m -> alpha = ln(2)/ln(10) + alpha = shear_exponent(_series([5.0]), _series([10.0])) + np.testing.assert_allclose(alpha.to_numpy(), np.log(2.0) / np.log(10.0)) + + def test_zero_shear(self) -> None: + alpha = shear_exponent(_series([8.0]), _series([8.0])) + np.testing.assert_allclose(alpha.to_numpy(), 0.0) + + def test_calm_is_nan(self) -> None: + alpha = shear_exponent(_series([0.0, -1.0, 5.0]), _series([5.0, 5.0, 0.0])) + assert alpha.isna().all() + + +class TestHubHeightWindSpeed: + def test_power_law_interpolation(self) -> None: + # alpha = 0.2, ws_100m = 10 -> ws_59m = 10 * 0.59^0.2 + ws = hub_height_wind_speed(_series([10.0]), _series([0.2]), hub_height_m=_HOT_HUB_HEIGHT_M) + np.testing.assert_allclose(ws.to_numpy(), 10.0 * 0.59**0.2) + + def test_zero_alpha_keeps_speed(self) -> None: + ws = hub_height_wind_speed(_series([7.0]), _series([0.0]), hub_height_m=_HOT_HUB_HEIGHT_M) + np.testing.assert_allclose(ws.to_numpy(), 7.0) + + def test_nan_alpha_propagates(self) -> None: + ws = hub_height_wind_speed(_series([7.0]), _series([np.nan]), hub_height_m=_HOT_HUB_HEIGHT_M) + assert ws.isna().all() + + +class TestGustRatio: + def test_known_value(self) -> None: + ratio = gust_ratio(_series([7.5]), _series([5.0])) + np.testing.assert_allclose(ratio.to_numpy(), 1.5) + + def test_calm_floor_is_nan(self) -> None: + ratio = gust_ratio(_series([3.0, 3.0]), _series([0.5, 0.0])) + assert ratio.isna().all() + + +class TestGustMargin: + def test_known_value(self) -> None: + margin = gust_margin(_series([12.0]), _series([9.0])) + np.testing.assert_allclose(margin.to_numpy(), 3.0) + + +class TestVerticalVeer: + @pytest.mark.parametrize( + ("wd_hi", "wd_lo", "expected"), + [ + (30.0, 10.0, 20.0), # simple positive veer + (10.0, 30.0, -20.0), # simple negative + (350.0, 10.0, -20.0), # wraps across north + (10.0, 350.0, 20.0), # wraps the other way + (180.0, 0.0, 180.0), # boundary maps to +180, not -180 + ], + ) + def test_wrapped_difference(self, wd_hi: float, wd_lo: float, expected: float) -> None: + veer = vertical_veer(_series([wd_hi]), _series([wd_lo])) + np.testing.assert_allclose(veer.to_numpy(), expected) + + +class TestAirDensity: + def test_standard_dry_air(self) -> None: + # ISA sea level: 15 degC, 1013.25 hPa, dry -> 1.225 kg/m3 + rho = air_density(_series([15.0]), _series([1013.25]), _series([0.0])) + np.testing.assert_allclose(rho.to_numpy(), 1.225, atol=0.001) + + def test_humidity_reduces_density(self) -> None: + dry = air_density(_series([15.0]), _series([1013.25]), _series([0.0])) + moist = air_density(_series([15.0]), _series([1013.25]), _series([100.0])) + assert float(moist.iloc[0]) < float(dry.iloc[0]) + np.testing.assert_allclose(moist.to_numpy(), 1.217, atol=0.002) + + +class TestEra5DerivedFrame: + def _aligned(self, n: int = 4) -> pd.DataFrame: + idx = pd.date_range("2020-01-01", periods=n, freq="10min", tz="UTC") + rng = np.random.default_rng(0) + return pd.DataFrame( + { + "wind_speed_10m": rng.uniform(3, 10, n), + "wind_speed_100m": rng.uniform(5, 14, n), + "wind_gusts_10m": rng.uniform(5, 15, n), + "wind_direction_10m": rng.uniform(0, 360, n), + "wind_direction_100m": rng.uniform(0, 360, n), + "temperature_2m": rng.uniform(0, 20, n), + "surface_pressure": rng.uniform(980, 1030, n), + "relative_humidity_2m": rng.uniform(40, 100, n), + }, + index=idx, + ) + + def test_all_derivations_present_and_finite(self) -> None: + frame = era5_derived_frame(self._aligned(), derivations=ERA5_DERIVATIONS, hub_height_m=_HOT_HUB_HEIGHT_M) + assert list(frame.columns) == list(ERA5_DERIVATIONS) + assert np.isfinite(frame.to_numpy(dtype=float)).all() + + def test_subset_only_builds_requested(self) -> None: + frame = era5_derived_frame(self._aligned(), derivations=["gust_ratio", "veer"]) + assert list(frame.columns) == ["gust_ratio", "veer"] + + def test_unknown_derivation_raises(self) -> None: + with pytest.raises(ValueError, match="unknown ERA5 derivation"): + era5_derived_frame(self._aligned(), derivations=["nonsense"]) + + def test_hub_height_required(self) -> None: + with pytest.raises(ValueError, match="requires hub_height_m"): + era5_derived_frame(self._aligned(), derivations=["wind_speed_hub"]) + + def test_missing_raw_column_raises(self) -> None: + aligned = self._aligned().drop(columns=["wind_gusts_10m"]) + with pytest.raises(ValueError, match="missing columns"): + era5_derived_frame(aligned, derivations=["gust_ratio"]) diff --git a/tests/benchmarking/baselines/test_hot_context.py b/tests/benchmarking/baselines/test_hot_context.py new file mode 100644 index 00000000..3765710f --- /dev/null +++ b/tests/benchmarking/baselines/test_hot_context.py @@ -0,0 +1,86 @@ +"""Offline tests for the vendored HoT assets and the v0 source-context builder.""" + +from __future__ import annotations + +import datetime as dt +from pathlib import Path + +import pandas as pd +import yaml + +from benchmarking.baselines import hot_context +from benchmarking.baselines.hot_context import ASSET_YAML, NORTHING_YAML, HotV0Context, build_hot_v0_context +from wind_up.yaml_loader import Loader, construct_include + + +def _load_yaml_with_includes(path): # noqa: ANN001, ANN202 + yaml.add_constructor("!include", construct_include, Loader) + with path.open() as f: + return yaml.load(f, Loader) # noqa: S506 + + +class TestVendoredAssets: + def test_asset_yaml_loads_with_turbine_type_include(self) -> None: + asset = _load_yaml_with_includes(ASSET_YAML) + assert asset["name"] == "Hill of Towie" + assert asset["wtgs"] == [f"T{i:02d}" for i in range(1, 22)] + assert asset["turbine_types"][0]["turbine_type"] == "SWT-2.3-82" + assert asset["turbine_types"][0]["rated_power_kw"] == 2300 + + def test_northing_yaml_covers_all_turbines(self) -> None: + with NORTHING_YAML.open() as f: + corrections = yaml.safe_load(f) + assert isinstance(corrections, list) + name, when, value = corrections[0] + assert name == "T01" + assert isinstance(when, dt.datetime) + assert isinstance(value, float) + assert {row[0] for row in corrections} == {f"T{i:02d}" for i in range(1, 22)} + + +class TestGetHotReanalysisDatasets: + def test_wraps_era5_df_for_hot_site(self, monkeypatch) -> None: # noqa: ANN001 + captured: dict[str, object] = {} + era5_df = pd.DataFrame({"wind_speed_100m": [1.0, 2.0]}) + + def fake_get_era5(**kwargs): # noqa: ANN003, ANN202 + captured.update(kwargs) + return era5_df + + monkeypatch.setattr(hot_context, "get_era5_hourly_df", fake_get_era5) + + datasets = hot_context.get_hot_reanalysis_datasets() + + assert len(datasets) == 1 + assert datasets[0].id == f"ERA5_{hot_context.HOT_LAT:.2f}_{hot_context.HOT_LON:.2f}" + pd.testing.assert_frame_equal(datasets[0].data, era5_df) + assert captured == { + "lat": hot_context.HOT_LAT, + "lon": hot_context.HOT_LON, + "start_date": hot_context.HOT_ERA5_START, + "end_date": hot_context.HOT_ERA5_END, + } + + +class TestBuildHotV0Context: + def test_wires_metadata_and_reanalysis_loaders(self, monkeypatch) -> None: # noqa: ANN001 + metadata = pd.DataFrame({"Name": ["T01"], "Latitude": [57.5], "Longitude": [-3.25]}) + sentinel_reanalysis = [object()] + captured: dict[str, object] = {} + + def fake_metadata(*, data_dir=None, wtg_names=None): # noqa: ANN001, ANN202 + captured["data_dir"] = data_dir + captured["wtg_names"] = wtg_names + return metadata + + monkeypatch.setattr(hot_context, "load_hot_metadata", fake_metadata) + monkeypatch.setattr(hot_context, "get_hot_reanalysis_datasets", lambda: sentinel_reanalysis) + + ctx = build_hot_v0_context(data_dir=Path("/some/dir")) + + assert isinstance(ctx, HotV0Context) + pd.testing.assert_frame_equal(ctx.metadata_df, metadata) + assert ctx.reanalysis_datasets is sentinel_reanalysis + assert ctx.asset_yaml == ASSET_YAML + assert ctx.northing_yaml == NORTHING_YAML + assert captured["data_dir"] == Path("/some/dir") diff --git a/tests/benchmarking/baselines/test_inspect_prepost_hard_case.py b/tests/benchmarking/baselines/test_inspect_prepost_hard_case.py new file mode 100644 index 00000000..c5689a1c --- /dev/null +++ b/tests/benchmarking/baselines/test_inspect_prepost_hard_case.py @@ -0,0 +1,23 @@ +"""The hard-case inspector's conditional-uplift plotting helper.""" + +from __future__ import annotations + +import pandas as pd + +from benchmarking.baselines.inspect_prepost_hard_case import conditional_truth_vs_estimate +from benchmarking.harness.method import MethodOutput + + +def test_conditional_truth_vs_estimate_merges_estimate_and_truth() -> None: + out = MethodOutput( + p50_overall=0.05, + p50_by_condition=pd.DataFrame( + {"condition": ["ws", "ws"], "condition_bin": ["(6.0, 8.0]", "(8.0, 10.0]"], "p50_uplift": [0.06, 0.04]} + ), + ) + truth = pd.DataFrame({"condition_bin": ["(6.0, 8.0]", "(8.0, 10.0]"], "true_uplift": [0.05, 0.05]}) + merged = conditional_truth_vs_estimate(out, {"ws": truth}, method_name="power_model") + row = merged[merged["condition_bin"] == "(6.0, 8.0]"].iloc[0] + assert row["mean_estimate"] == 0.06 + assert row["mean_truth"] == 0.05 + assert row["method"] == "power_model" diff --git a/tests/benchmarking/baselines/test_naive_ratio.py b/tests/benchmarking/baselines/test_naive_ratio.py new file mode 100644 index 00000000..493661e2 --- /dev/null +++ b/tests/benchmarking/baselines/test_naive_ratio.py @@ -0,0 +1,442 @@ +"""Tests for the NaiveRatioMethod energy-ratio baseline. + +The method is light (no wind_up pipeline), so these run the real thing on small +hand-built SCADA frames. They cover the estimator (prepost + toggle recovery), +complete-case data handling, the timebase variable, the active-power-only rule, the +written diagnostics, and the error/degenerate paths. +""" + +from __future__ import annotations + +import ast +from dataclasses import replace +from pathlib import Path + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines import naive_ratio +from benchmarking.baselines.naive_ratio import ( + NaiveRatioMethod, + _daily_segment_coverage, + _daily_segment_ratio, + _expected_per_day, + _infer_timebase, +) +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.synthetic import ColumnSchema, ToggleSchedule, treated_mask + +# Deliberately non-v0 column names: the method is source-agnostic, so the active-power column +# is configured and the turbine column comes from the seam. Using names that are nothing like +# v0's ``DataColumns`` proves the method never reaches for wind_up's vocabulary. +_TURBINE_COL = "asset_id" +_POWER_COL = "kw" +_AVAIL_COL = "secs_avail" +# Larger than any test timebase's full period, so the (now required) availability filter keeps +# every row unless a test deliberately sets a lower value. +_FULLY_AVAILABLE_SECS = 3600.0 +# The method reads active_power + availability from the schema; the other required roles are +# unused by the naive ratio, so name them with placeholders. +_COLUMNS = ColumnSchema( + turbine=_TURBINE_COL, + active_power=_POWER_COL, + wind_speed="ws", + wind_speed_sd="ws_sd", + gen_rpm="rpm", + availability=_AVAIL_COL, +) + + +def _index(n: int, *, freq: str = "10min", start: str = "2020-01-01") -> pd.DatetimeIndex: + return pd.date_range(start=start, periods=n, freq=freq, tz="UTC", name="timestamp") + + +def _scada(power_by_turbine: dict[str, np.ndarray], index: pd.DatetimeIndex) -> pd.DataFrame: + """Long-format SCADA: one block of rows per turbine, sharing ``index`` (fully available).""" + frames = [ + pd.DataFrame( + {_TURBINE_COL: name, _POWER_COL: np.asarray(vals, dtype=float), _AVAIL_COL: _FULLY_AVAILABLE_SECS}, + index=index, + ) + for name, vals in power_by_turbine.items() + ] + return pd.concat(frames) + + +def _recovery_scada( + index: pd.DatetimeIndex, *, treated: np.ndarray, k: float = 0.8, uplift: float = 0.05 +) -> pd.DataFrame: + """Build SCADA where test = k*ref_total in baseline and (1+uplift)*k*ref_total when treated.""" + ref1 = np.linspace(100.0, 1000.0, len(index)) + ref2 = np.linspace(50.0, 500.0, len(index)) + ref_total = ref1 + ref2 + test = k * ref_total + test = np.where(treated, test * (1.0 + uplift), test) + return _scada({"T1": test, "R1": ref1, "R2": ref2}, index) + + +class TestInferTimebase: + def test_infers_ten_minutes(self) -> None: + assert _infer_timebase(_index(20)) == pd.Timedelta(minutes=10) + + def test_infers_thirty_minutes(self) -> None: + assert _infer_timebase(_index(20, freq="30min")) == pd.Timedelta(minutes=30) + + def test_infers_from_duplicated_long_index(self) -> None: + # long format repeats each timestamp once per turbine + idx = _index(10) + doubled = idx.append(idx) + assert _infer_timebase(doubled) == pd.Timedelta(minutes=10) + + +def test_naive_ratio_shares_no_wind_up_code() -> None: + """The naive method is an independent, source-native baseline: it must not import wind_up.""" + tree = ast.parse(Path(naive_ratio.__file__).read_text()) + modules = {alias.name for node in ast.walk(tree) if isinstance(node, ast.Import) for alias in node.names} + modules |= {node.module for node in ast.walk(tree) if isinstance(node, ast.ImportFrom) and node.module} + offenders = {m for m in modules if m == "wind_up" or m.startswith("wind_up.")} + assert not offenders, f"naive_ratio must not depend on wind_up, found imports: {offenders}" + + +class TestRecovery: + def test_prepost_recovers_known_uplift(self) -> None: + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.07) + out = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + assert isinstance(out, MethodOutput) + assert out.p50_overall == pytest.approx(0.07) + assert out.p50_by_condition is None + + def test_toggle_recovers_known_uplift(self) -> None: + idx = _index(40) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=idx[0]) + treated = treated_mask(idx, schedule) + scada = _recovery_scada(idx, treated=treated, uplift=0.03) + out = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_overall == pytest.approx(0.03) + + +class TestDowntimeFilter: + """Downtime filtering is required and applies to the test turbine and every reference.""" + + def test_columns_is_required(self) -> None: + with pytest.raises(TypeError): + NaiveRatioMethod() # type: ignore[call-arg] + + @pytest.mark.parametrize("blank", ["", " "]) + def test_blank_availability_role_raises(self, blank: str) -> None: + # A schema that leaves the availability role blank (empty or whitespace) would silently skip + # downtime filtering, so construction must reject it. + with pytest.raises(ValueError, match="availability"): + NaiveRatioMethod(columns=replace(_COLUMNS, availability=blank)) + + def test_missing_availability_column_raises(self) -> None: + idx = _index(20) + upgrade = idx[10] + scada = _recovery_scada(idx, treated=np.asarray(idx >= upgrade), uplift=0.05).drop(columns=[_AVAIL_COL]) + with pytest.raises(ValueError, match="availability"): + NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + + def test_down_reference_timestamps_are_excluded(self) -> None: + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + # Corrupt a reference's power at two baseline timestamps AND mark it unavailable there. The + # downtime filter must drop those timestamps, so the wild power never biases the ratio. + corrupted = scada.copy() + down = (corrupted[_TURBINE_COL] == "R1") & corrupted.index.isin([idx[3], idx[4]]) + corrupted.loc[down, _POWER_COL] = 1e6 + corrupted.loc[down, _AVAIL_COL] = 0.0 + out = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=corrupted, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + assert out.p50_overall == pytest.approx(0.05) + + +class TestCompleteCase: + def test_drops_timestamp_when_a_reference_is_nan(self) -> None: + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + clean = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + + # NaN one reference at a used baseline timestamp: that timestamp must be excluded, but + # since the ratio is identical at every row the estimate is unchanged. + corrupted = scada.copy() + mask = (corrupted[_TURBINE_COL] == "R1") & (corrupted.index == idx[3]) + corrupted.loc[mask, _POWER_COL] = np.nan + out = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=corrupted, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + assert out.p50_overall == pytest.approx(clean.p50_overall) + + def test_nan_at_unused_timestamp_does_not_change_estimate(self) -> None: + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + # make idx[2] already unused (test NaN), then add a second NaN at the same timestamp + base = scada.copy() + base.loc[(base[_TURBINE_COL] == "T1") & (base.index == idx[2]), _POWER_COL] = np.nan + before = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=base, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + after_df = base.copy() + after_df.loc[(after_df[_TURBINE_COL] == "R1") & (after_df.index == idx[2]), _POWER_COL] = np.nan + after = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=after_df, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + assert after.p50_overall == pytest.approx(before.p50_overall) + + +class TestTimebaseInvariance: + def test_estimate_invariant_to_timebase(self) -> None: + treated10 = None + results = [] + for freq in ("10min", "30min"): + idx = _index(20, freq=freq) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + if treated10 is None: + treated10 = treated + scada = _recovery_scada(idx, treated=treated, uplift=0.04) + results.append( + NaiveRatioMethod(columns=_COLUMNS) + .estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL)) + .p50_overall + ) + assert results[0] == pytest.approx(results[1]) + + def test_override_timebase_used_for_mwh(self, tmp_path) -> None: # noqa: ANN001 + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + method = NaiveRatioMethod( + columns=_COLUMNS, + out_dir=tmp_path, + timebase=pd.Timedelta(minutes=30), + ) + method.estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL)) + stats = _read_only_csv(tmp_path, "data_stats") + all_row = stats[stats["segment"] == "all"].iloc[0] + # MWh = sum(power_kw) * timebase_hours / 1000; check it used 0.5h not 1/6h + expected = all_row["used_test_mean_power_kw"] * all_row["n_used_timestamps"] * 0.5 / 1000.0 + assert all_row["used_test_mwh"] == pytest.approx(expected) + + +class TestActivePowerOnly: + def test_extra_columns_ignored(self) -> None: + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + plain = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + + with_extra = scada.copy() + rng = np.random.default_rng(0) + with_extra["ws"] = rng.normal(size=len(with_extra)) + with_extra["rpm"] = rng.normal(size=len(with_extra)) + out = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=with_extra, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + assert out.p50_overall == pytest.approx(plain.p50_overall) + + +class TestErrors: + def test_no_reference_raises(self) -> None: + idx = _index(20) + upgrade = idx[10] + scada = _scada({"T1": np.ones(len(idx))}, idx) + with pytest.raises(ValueError, match="reference"): + NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + + def test_no_used_baseline_returns_nan(self) -> None: + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + # NaN the test turbine across the whole baseline -> no used baseline timestamps + is_baseline_test = (scada[_TURBINE_COL] == "T1") & np.asarray(scada.index < upgrade) + scada.loc[is_baseline_test, _POWER_COL] = np.nan + out = NaiveRatioMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + assert np.isnan(out.p50_overall) + + +class TestPlotSeries: + """The per-segment daily series that back the revised ratio plot and the new coverage plot.""" + + def test_ratio_series_split_by_segment(self) -> None: + # one day baseline (ratio 0.8) then one day upgraded (ratio 0.8*1.05); daily sum-based ratio. + idx = _index(288, freq="10min") # two full days + upgrade = idx[144] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + wide = scada.pivot_table(index=scada.index, columns=_TURBINE_COL, values=_POWER_COL) + test_pw = wide["T1"].to_numpy() + ref_total = wide[["R1", "R2"]].sum(axis=1).to_numpy() + used = np.ones(len(wide), dtype=bool) + + base = _daily_segment_ratio(wide.index, test_pw, ref_total, used & ~treated) + up = _daily_segment_ratio(wide.index, test_pw, ref_total, used & treated) + # each segment only has data on its own day; the other day is NaN. + assert base.dropna().to_numpy() == pytest.approx(0.8) + assert up.dropna().to_numpy() == pytest.approx(0.8 * 1.05) + assert base.index.equals(up.index) + + def test_prepost_coverage_is_per_day_share_of_expected(self) -> None: + # day 1 entirely baseline, day 2 entirely upgraded (prepost). Coverage is the segment's + # share of the day's 144 expected timestamps, so the inactive segment reads 0 that day. + idx = _index(288, freq="10min") + upgrade = idx[144] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + wide = scada.pivot_table(index=scada.index, columns=_TURBINE_COL, values=_POWER_COL) + used = np.ones(len(wide), dtype=bool) + used[10] = used[200] = False # drop one timestamp in each day + + expected = _expected_per_day(wide.index, _infer_timebase(wide.index)) + assert expected.to_numpy() == pytest.approx(144) + base = _daily_segment_coverage(wide.index, used, ~treated, expected) + up = _daily_segment_coverage(wide.index, used, treated, expected) + for series in (base, up): + assert ((series >= 0.0) & (series <= 1.0)).all() + # baseline: 143/144 on day 1, 0 on day 2 (no baseline data); upgraded mirrors it. + assert base.to_numpy() == pytest.approx([143 / 144, 0.0]) + assert up.to_numpy() == pytest.approx([0.0, 143 / 144]) + # the two segments sum to the day's overall complete-case coverage. + assert (base + up).to_numpy() == pytest.approx([143 / 144, 143 / 144]) + + def test_toggle_coverage_capped_near_duty_cycle(self) -> None: + # 20-on/20-off toggle on 10-min data -> each segment can occupy at most ~50% of a day. + idx = _index(288, freq="10min") + schedule = ToggleSchedule(period=pd.Timedelta(minutes=40), start=idx[0]) + treated = np.asarray(treated_mask(idx, schedule)) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + wide = scada.pivot_table(index=scada.index, columns=_TURBINE_COL, values=_POWER_COL) + used = np.ones(len(wide), dtype=bool) + + expected = _expected_per_day(wide.index, _infer_timebase(wide.index)) + base = _daily_segment_coverage(wide.index, used, ~treated, expected) + up = _daily_segment_coverage(wide.index, used, treated, expected) + assert base.to_numpy() == pytest.approx(0.5) + assert up.to_numpy() == pytest.approx(0.5) + + +def _read_only_csv(folder: Path, kind: str) -> pd.DataFrame: + run_dirs = [p for p in Path(folder).iterdir() if p.is_dir()] + assert len(run_dirs) == 1, f"expected exactly one run dir, found {run_dirs}" + matches = list(run_dirs[0].glob(f"*_{kind}_*.csv")) + assert len(matches) == 1, f"expected one {kind} csv, found {matches}" + return pd.read_csv(matches[0]) + + +class TestDiagnostics: + def _run(self, tmp_path, *, save_plots: bool = False): # noqa: ANN001, ANN202 + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.06) + method = NaiveRatioMethod(columns=_COLUMNS, out_dir=tmp_path, save_plots=save_plots) + out = method.estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + return out, idx, upgrade + + def test_writes_both_csvs(self, tmp_path) -> None: # noqa: ANN001 + self._run(tmp_path) + stats = _read_only_csv(tmp_path, "data_stats") + results = _read_only_csv(tmp_path, "results") + assert sorted(stats["segment"]) == ["all", "baseline", "upgraded"] + assert len(results) == 1 + + def test_uplift_rederivable_from_stats(self, tmp_path) -> None: # noqa: ANN001 + out, _, _ = self._run(tmp_path) + stats = _read_only_csv(tmp_path, "data_stats").set_index("segment") + rho_base = stats.loc["baseline", "used_test_mwh"] / stats.loc["baseline", "used_ref_total_mwh"] + rho_up = stats.loc["upgraded", "used_test_mwh"] / stats.loc["upgraded", "used_ref_total_mwh"] + assert (rho_up / rho_base - 1.0) == pytest.approx(out.p50_overall) + + def test_results_csv_matches_estimate(self, tmp_path) -> None: # noqa: ANN001 + out, _, _ = self._run(tmp_path) + results = _read_only_csv(tmp_path, "results").iloc[0] + assert results["uplift_frc"] == pytest.approx(out.p50_overall) + assert results["mode"] == "prepost" + assert results["n_refs"] == 2 + + def test_stats_coverage_full_for_complete_data(self, tmp_path) -> None: # noqa: ANN001 + self._run(tmp_path) + stats = _read_only_csv(tmp_path, "data_stats").set_index("segment") + assert stats.loc["all", "rows_data_coverage"] == pytest.approx(1.0) + assert stats.loc["all", "used_data_coverage"] == pytest.approx(1.0) + assert stats.loc["all", "n_used_timestamps"] == 20 + + def test_save_plots_writes_pngs(self, tmp_path) -> None: # noqa: ANN001 + self._run(tmp_path, save_plots=True) + run_dir = next(p for p in Path(tmp_path).iterdir() if p.is_dir()) + names = {p.name for p in (run_dir / "plots").rglob("*.png")} + # the method's own naive plots, plus the shared cross-method diagnostics it now emits + assert {"T1_scatter.png", "T1_ratio_timeseries.png", "T1_coverage_timeseries.png"} <= names + # a run-config YAML is written alongside the plots + assert any(run_dir.glob("config_*.yaml")) + + def test_no_plots_by_default(self, tmp_path) -> None: # noqa: ANN001 + self._run(tmp_path) + run_dir = next(p for p in Path(tmp_path).iterdir() if p.is_dir()) + assert not (run_dir / "plots").exists() + + +class TestToggleCampaignOnly: + """``toggle_campaign_only`` restricts a toggle fit to the campaign window (drops pre-campaign).""" + + def _toggle_scada(self) -> tuple[pd.DataFrame, ToggleSchedule, pd.Timestamp]: + idx = _index(300) + start = idx[100] # 100 pre-campaign rows, then 200 rows of interleaved on/off + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=start) + treated = np.asarray(treated_mask(idx, schedule)) + return _recovery_scada(idx, treated=treated, uplift=0.05), schedule, start + + def test_campaign_only_excludes_precampaign_from_baseline(self, tmp_path) -> None: # noqa: ANN001 + scada, schedule, start = self._toggle_scada() + NaiveRatioMethod(columns=_COLUMNS, out_dir=tmp_path / "on").estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + NaiveRatioMethod( + columns=_COLUMNS, + out_dir=tmp_path / "off", + toggle_campaign_only=False, + ).estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL)) + base_on = _read_only_csv(tmp_path / "on", "data_stats").set_index("segment").loc["baseline"] + base_off = _read_only_csv(tmp_path / "off", "data_stats").set_index("segment").loc["baseline"] + # campaign-only drops the 100 pre-campaign rows from the baseline class + assert base_on["n_used_timestamps"] < base_off["n_used_timestamps"] + assert pd.Timestamp(base_on["first_timestamp"]) >= start + + def test_prepost_unaffected_by_flag(self) -> None: + idx = _index(20) + upgrade = idx[10] + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + a = NaiveRatioMethod(columns=_COLUMNS).estimate(mi) + b = NaiveRatioMethod(columns=_COLUMNS, toggle_campaign_only=False).estimate(mi) + assert a.p50_overall == pytest.approx(b.p50_overall) diff --git a/tests/benchmarking/baselines/test_power_model_conditional.py b/tests/benchmarking/baselines/test_power_model_conditional.py new file mode 100644 index 00000000..6d9e31f9 --- /dev/null +++ b/tests/benchmarking/baselines/test_power_model_conditional.py @@ -0,0 +1,93 @@ +"""Unit tests for the pure conditional-decomposition helpers (imputation + energy-identity re-level).""" + +from __future__ import annotations + +import numpy as np +import pytest + +from benchmarking.baselines.power_model.conditional import impute_uncovered_bins, relevel_conditional + + +class TestImputeUncoveredBins: + def test_ws_bfills_low_holes_from_the_nearest_covered_bin_above(self) -> None: + one_plus_u = np.array([np.nan, np.nan, 1.08, 1.06, np.nan]) + measured = np.array([False, False, True, True, False]) + out = impute_uncovered_bins(one_plus_u, condition="ws", measured=measured, one_plus_overall=1.03) + # low holes take the nearest covered bin above; trailing hole -> 1.0 (0 uplift at rated) + assert out.tolist() == pytest.approx([1.08, 1.08, 1.08, 1.06, 1.0]) + + def test_ws_all_uncovered_top_fills_to_zero_uplift(self) -> None: + one_plus_u = np.array([np.nan, np.nan]) + measured = np.array([False, False]) + out = impute_uncovered_bins(one_plus_u, condition="ws", measured=measured, one_plus_overall=1.03) + assert out.tolist() == pytest.approx([1.0, 1.0]) + + def test_ti_fills_uncovered_at_overall(self) -> None: + one_plus_u = np.array([1.07, np.nan, np.nan]) + measured = np.array([True, False, False]) + out = impute_uncovered_bins(one_plus_u, condition="ti", measured=measured, one_plus_overall=1.03) + assert out.tolist() == pytest.approx([1.07, 1.03, 1.03]) + + def test_measured_bins_pass_through_even_if_shape_is_finite_elsewhere(self) -> None: + one_plus_u = np.array([1.20, 1.05, 1.10]) + measured = np.array([True, True, True]) + out = impute_uncovered_bins(one_plus_u, condition="ws", measured=measured, one_plus_overall=1.03) + assert out.tolist() == pytest.approx([1.20, 1.05, 1.10]) + + def test_power_behaves_like_ws_bfill_then_zero_at_rated(self) -> None: + # power is monotone-saturating like ws: low holes bfill from the covered bin above, and the + # trailing (near-rated) hole takes 1.0 (0 uplift at rated). + one_plus_u = np.array([np.nan, 1.08, 1.06, np.nan]) + measured = np.array([False, True, True, False]) + out = impute_uncovered_bins(one_plus_u, condition="power", measured=measured, one_plus_overall=1.03) + assert out.tolist() == pytest.approx([1.08, 1.08, 1.06, 1.0]) + + def test_unknown_condition_raises(self) -> None: + with pytest.raises(ValueError, match="unknown condition"): + impute_uncovered_bins(np.array([1.1]), condition="wd", measured=np.array([True]), one_plus_overall=1.0) + + def test_measured_bin_with_nan_shape_is_not_trusted(self) -> None: + # a bin flagged measured but carrying a NaN shape (shouldn't happen, but be defensive) is filled + one_plus_u = np.array([1.05, np.nan, 1.02]) + measured = np.array([True, True, True]) + out = impute_uncovered_bins(one_plus_u, condition="ti", measured=measured, one_plus_overall=1.04) + assert out.tolist() == pytest.approx([1.05, 1.04, 1.02]) + + +def _agg(sum_actual: np.ndarray, one_plus_u: np.ndarray) -> float: + """Energy-weighted aggregation of a per-bin (1+u): ratio-of-sums Σactual / Σ(actual/(1+u)).""" + a = np.asarray(sum_actual, float) + u = np.asarray(one_plus_u, float) + return float(a.sum() / (a / u).sum()) + + +class TestRelevelConditionalPinned: + def test_measured_and_pinned_imputed_aggregate_to_overall(self) -> None: + sum_actual = np.array([1000.0, 2000.0, 500.0]) + one_plus_u = np.array([1.10, 0.95, 1.02]) # last is an imputed pin + measured = np.array([True, True, False]) + final = relevel_conditional(sum_actual, one_plus_u, measured=measured, one_plus_overall=1.05) + assert final[2] == pytest.approx(1.02) # imputed bin unchanged (pinned) + assert _agg(sum_actual, final) == pytest.approx(1.05) # whole thing aggregates to overall + + def test_all_measured_reduces_to_a_single_scale(self) -> None: + sum_actual = np.array([1000.0, 1000.0]) + one_plus_u = np.array([1.1, 1.1]) + measured = np.array([True, True]) + final = relevel_conditional(sum_actual, one_plus_u, measured=measured, one_plus_overall=1.1) + assert final == pytest.approx(one_plus_u) + + def test_no_measured_bins_falls_back_to_overall_everywhere(self) -> None: + sum_actual = np.array([1000.0, 500.0]) + one_plus_u = np.array([1.5, 0.7]) # imputed pins only + measured = np.array([False, False]) + final = relevel_conditional(sum_actual, one_plus_u, measured=measured, one_plus_overall=1.04) + assert final.tolist() == pytest.approx([1.04, 1.04]) + + def test_nonpositive_denominator_falls_back_to_overall(self) -> None: + # imputed counterfactual energy already exceeds the headline total -> cannot solve λ>0 + sum_actual = np.array([1000.0, 1000.0]) + one_plus_u = np.array([1.10, 0.20]) # bin 1 imputed with a huge implied cf + measured = np.array([True, False]) + final = relevel_conditional(sum_actual, one_plus_u, measured=measured, one_plus_overall=1.5) + assert final.tolist() == pytest.approx([1.5, 1.5]) diff --git a/tests/benchmarking/baselines/test_power_model_diagnostics.py b/tests/benchmarking/baselines/test_power_model_diagnostics.py new file mode 100644 index 00000000..49b38ffd --- /dev/null +++ b/tests/benchmarking/baselines/test_power_model_diagnostics.py @@ -0,0 +1,180 @@ +"""Tests for the power-model residual diagnostics (the shrinkage-check plot).""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd + +from benchmarking.baselines.power_model.diagnostics import ( + DiagnosticData, + _as_percent_of_power, + _binned_stats, + _condition_diagnostic_figure, + _plot_residual_binned, + _set_ylim_from_inliers, + plot_conditional_diagnostics, +) +from benchmarking.diagnostics import stages + + +def _toy_per_bin() -> pd.DataFrame: + """A per-bin conditional frame spanning ws, ti and power with a covered/imputed mix.""" + rows = [] + specs = { + "ws": ["(4.0, 6.0]", "(6.0, 8.0]", "(8.0, 10.0]"], + "ti": ["(0.05, 0.1]", "(0.1, 0.15]"], + "power": ["(230.0, 690.0]", "(690.0, 1150.0]"], + } + for cond, bins in specs.items(): + for i, b in enumerate(bins): + rows.append( + { + "condition": cond, + "condition_bin": b, + "r_fwd": 0.05 + 0.01 * i, + "r_rev": -0.04 + 0.01 * i, + "implied_shrinkage": 0.98 + 0.01 * i, + "p50_uplift": 0.05 + 0.005 * i, + "covered": i % 2 == 0, # alternate covered / imputed + } + ) + return pd.DataFrame(rows) + + +def test_plot_conditional_diagnostics_writes_a_figure_per_condition(tmp_path: Path) -> None: + plot_conditional_diagnostics(tmp_path, _toy_per_bin(), test_wtg="T07") + for cond in ("ws", "ti", "power"): + assert (tmp_path / f"conditional_{cond}.png").exists() + + +def test_condition_diagnostic_figure_has_four_panels() -> None: + sub = _toy_per_bin().query("condition == 'power'") + fig = _condition_diagnostic_figure(sub, condition="power", test_wtg="T07") + assert len(fig.axes) == 4 + plt.close(fig) + + +def test_condition_diagnostic_figure_shades_covered_and_imputed_distinctly() -> None: + # the uplift panel must visually separate measured (covered) bins from imputed ones + sub = _toy_per_bin().query("condition == 'ws'") # covered = [True, False, True] + fig = _condition_diagnostic_figure(sub, condition="ws", test_wtg="T07") + uplift_ax = fig.axes[2] # panels: fwd, rev, uplift, shrinkage + facecolors = {tuple(patch.get_facecolor()) for patch in uplift_ax.patches} + assert len(facecolors) >= 2 # covered and imputed bins are not the same colour + plt.close(fig) + + +if TYPE_CHECKING: + from pathlib import Path + + +def test_binned_stats_mean_sd_and_count() -> None: + x = np.array([0.5, 1.5, 1.6, 1.7, 10.0]) # last value falls outside the edges + y = np.array([1.0, 10.0, 12.0, 14.0, 999.0]) + edges = np.array([0.0, 1.0, 2.0]) + centers, mean, sd, count = _binned_stats(x, y, edges) + assert list(centers) == [0.5, 1.5] + # bin [0,1): a single point -> below _MIN_BIN_COUNT, so NaN mean/SD but count recorded + assert count[0] == 1 + assert np.isnan(mean[0]) + assert np.isnan(sd[0]) + # bin [1,2): three points {10,12,14} -> mean 12, sample SD 2 + assert count[1] == 3 + assert mean[1] == 12.0 + assert sd[1] == 2.0 + + +def test_binned_stats_all_nan_input_is_safe() -> None: + edges = np.array([0.0, 1.0, 2.0]) + centers, mean, sd, count = _binned_stats(np.full(3, np.nan), np.arange(3.0), edges) + assert len(centers) == 2 + assert np.isnan(mean).all() + assert np.isnan(sd).all() + assert (count == 0).all() + + +def _diag_data(*, with_conditions: bool) -> DiagnosticData: + """A minimal DiagnosticData carrying only what the residual-binned plot reads.""" + rng = np.random.default_rng(0) + n = 400 + y_base = rng.uniform(0, 2000, n) + pred_base = 0.7 * y_base + 300 # deliberate shrinkage: slope < 1 + y_up = rng.uniform(0, 2000, n) + pred_up = 0.7 * y_up + 300 + cond_up = cond_base = None + if with_conditions: + cond_base = pd.DataFrame({"ws": rng.uniform(0, 25, n), "ti": rng.uniform(0, 0.4, n)}) + cond_up = pd.DataFrame({"ws": rng.uniform(0, 25, n), "ti": rng.uniform(0, 0.4, n)}) + return DiagnosticData( + test_wtg="T07", + mode="prepost", + index=pd.DatetimeIndex([]), + treated_all=np.array([]), + selected_all=np.array([]), + y_all=np.array([]), + timebase=pd.Timedelta(minutes=10), + upgraded_ts=pd.DatetimeIndex([]), + y_upgraded=y_up, + pred_upgraded=pred_up, + y_baseline_valid=y_base, + pred_baseline_valid=pred_base, + feature_names=[], + feature_values=pd.DataFrame(), + y_selected=np.array([]), + outcome_model=None, + overall_uplift=0.0, + sum_actual_kw=0.0, + sum_counterfactual_kw=0.0, + n_refs=3, + era5_lag_rows=None, + era5_corr=None, + era5_sweep=None, + cond_upgraded=cond_up, + cond_baseline_valid=cond_base, + ) + + +def test_as_percent_of_power_divides_per_bin_and_drops_nonpositive() -> None: + out = _as_percent_of_power(np.array([10.0, 5.0, -3.0]), np.array([100.0, 0.0, 60.0])) + assert out[0] == 10.0 # 10 kW of 100 kW + assert np.isnan(out[1]) # mean power 0 -> dropped + assert out[2] == -5.0 # -3 kW of 60 kW + + +def test_set_ylim_from_inliers_ignores_out_of_range_points() -> None: + _, ax = plt.subplots() + # inliers within +/-30 are {-10, 20}; the -330 outlier must not stretch the limits + _set_ylim_from_inliers(ax, [np.array([-10.0, 20.0, -330.0, np.nan])]) + lo, hi = ax.get_ylim() + assert lo < -10.0 # a small margin below the min inlier + assert lo > -20.0 # but nowhere near the -330 outlier + assert 20.0 < hi < 30.0 + plt.close() + + +def test_set_ylim_from_inliers_noop_when_no_inliers() -> None: + _, ax = plt.subplots() + before = ax.get_ylim() + _set_ylim_from_inliers(ax, [np.array([100.0, -330.0])]) # all outside +/-30 + assert ax.get_ylim() == before + plt.close() + + +def test_plot_residual_binned_writes_both_png_with_conditions(tmp_path: Path) -> None: + model_dir = tmp_path / stages.UPLIFT_MODELLING + model_dir.mkdir() + _plot_residual_binned(model_dir, _diag_data(with_conditions=True)) + assert (model_dir / "residual_binned.png").exists() + assert (model_dir / "residual_binned_pct.png").exists() + + +def test_plot_residual_binned_writes_png_without_conditions(tmp_path: Path) -> None: + # No ws/TI columns configured: the plot still renders the power-axis panels. + model_dir = tmp_path / stages.UPLIFT_MODELLING + model_dir.mkdir() + _plot_residual_binned(model_dir, _diag_data(with_conditions=False)) + assert (model_dir / "residual_binned.png").exists() + assert (model_dir / "residual_binned_pct.png").exists() diff --git a/tests/benchmarking/baselines/test_power_model_features.py b/tests/benchmarking/baselines/test_power_model_features.py new file mode 100644 index 00000000..80405f7f --- /dev/null +++ b/tests/benchmarking/baselines/test_power_model_features.py @@ -0,0 +1,184 @@ +"""Tests for the power-model curated, reference-only feature builder. + +Covers the reference feature pivot (active power + availability per reference, original tag names, +NaN-preserving), the ERA5 all-columns passthrough with direction sin/cos, the outcome extraction, +and the reference-only enforcement guard. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.era5_sync import ERA5_WD, ERA5_WS +from benchmarking.baselines.power_model.features import ( + QUALIFIER, + build_reference_features, + check_reference_only, + era5_feature_frame, + extract_outcome, +) + +_TURBINE = "TurbineName" +_POWER = "wtc_ActPower_mean" +_AVAIL = "wtc_ScReToOp_timeon" +_WS = "wtc_AcWindSp_mean" + + +def _index(n: int) -> pd.DatetimeIndex: + return pd.date_range("2020-01-01", periods=n, freq="10min", tz="UTC", name="timestamp") + + +def _scada(idx: pd.DatetimeIndex) -> pd.DataFrame: + """Long SCADA with test T1 and references R1, R2, R3, each carrying power + availability + ws.""" + rng = np.random.default_rng(0) + frames = [ + pd.DataFrame( + { + _TURBINE: name, + _POWER: rng.normal(800, 100, len(idx)), + _AVAIL: 600.0, + _WS: rng.normal(8, 2, len(idx)), + }, + index=idx, + ) + for name in ("T1", "R1", "R2", "R3") + ] + return pd.concat(frames) + + +class TestBuildReferenceFeatures: + def test_only_active_power_and_availability_per_reference(self) -> None: + idx = _index(12) + feats = build_reference_features( + _scada(idx), test_wtg="T1", turbine_col=_TURBINE, active_power_col=_POWER, availability_col=_AVAIL + ) + # three references x two value columns = six feature columns; none for T1; ws not included + assert len(feats.columns) == 6 + assert not any(c.endswith(f"{QUALIFIER}T1") for c in feats.columns) + assert not any(c.startswith(_WS) for c in feats.columns) + assert feats.index.equals(idx) + + def test_keeps_original_tag_names(self) -> None: + idx = _index(12) + feats = build_reference_features( + _scada(idx), test_wtg="T1", turbine_col=_TURBINE, active_power_col=_POWER, availability_col=_AVAIL + ) + assert f"{_POWER}{QUALIFIER}R1" in feats.columns + assert f"{_AVAIL}{QUALIFIER}R3" in feats.columns + + def test_preserves_nan_rows(self) -> None: + idx = _index(12) + scada = _scada(idx) + scada.loc[(scada[_TURBINE] == "R1") & (scada.index == idx[3]), _POWER] = np.nan + feats = build_reference_features( + scada, test_wtg="T1", turbine_col=_TURBINE, active_power_col=_POWER, availability_col=_AVAIL + ) + assert len(feats) == len(idx) + assert np.isnan(feats.loc[idx[3], f"{_POWER}{QUALIFIER}R1"]) + + def test_raises_when_no_references(self) -> None: + idx = _index(12) + only_test = _scada(idx) + only_test = only_test[only_test[_TURBINE] == "T1"] + with pytest.raises(ValueError, match="reference"): + build_reference_features( + only_test, test_wtg="T1", turbine_col=_TURBINE, active_power_col=_POWER, availability_col=_AVAIL + ) + + def test_extra_cols_add_per_reference_features(self) -> None: + idx = _index(12) + scada = _scada(idx) + scada["wtc_ActPower_stddev"] = 7.0 + feats = build_reference_features( + scada, + test_wtg="T1", + turbine_col=_TURBINE, + active_power_col=_POWER, + availability_col=_AVAIL, + extra_cols=("wtc_ActPower_stddev",), + ) + # three references x three value columns; still nothing from the test turbine + assert len(feats.columns) == 9 + assert f"wtc_ActPower_stddev{QUALIFIER}R2" in feats.columns + assert not any(c.endswith(f"{QUALIFIER}T1") for c in feats.columns) + + def test_missing_availability_col_raises_even_when_not_featured(self) -> None: + idx = _index(12) + scada = _scada(idx).drop(columns=[_AVAIL]) + with pytest.raises(ValueError, match="missing required reference-feature columns"): + build_reference_features( + scada, + test_wtg="T1", + turbine_col=_TURBINE, + active_power_col=_POWER, + availability_col=_AVAIL, + include_availability=False, + ) + + def test_missing_extra_col_raises(self) -> None: + idx = _index(12) + with pytest.raises(ValueError, match="missing required reference-feature columns"): + build_reference_features( + _scada(idx), + test_wtg="T1", + turbine_col=_TURBINE, + active_power_col=_POWER, + availability_col=_AVAIL, + extra_cols=("wtc_ActPower_stddev",), + ) + + def test_extra_test_turbine_column_never_reaches_features(self) -> None: + idx = _index(12) + scada = _scada(idx) + # a leak-bait column only present on the test turbine must not appear among features + scada["wtc_NacWdSp_mean"] = np.where(scada[_TURBINE] == "T1", scada[_POWER], np.nan) + feats = build_reference_features( + scada, test_wtg="T1", turbine_col=_TURBINE, active_power_col=_POWER, availability_col=_AVAIL + ) + assert not any("NacWdSp" in c for c in feats.columns) + assert not any(c.endswith(f"{QUALIFIER}T1") for c in feats.columns) + + +class TestEra5FeatureFrame: + def test_passes_through_raw_drops_aliases_and_adds_direction_sin_cos(self) -> None: + idx = _index(6) + aligned = pd.DataFrame( + { + "wind_speed_100m": np.linspace(5, 10, 6), + "wind_direction_100m": np.linspace(0, 270, 6), + "temperature_2m": np.linspace(1, 6, 6), + ERA5_WS: np.linspace(5, 10, 6), + ERA5_WD: np.linspace(0, 270, 6), + }, + index=idx, + ) + out = era5_feature_frame(aligned) + # aliases dropped; raw columns kept; direction gains sin/cos companions (raw degrees kept) + assert ERA5_WS not in out.columns + assert ERA5_WD not in out.columns + assert {"wind_speed_100m", "wind_direction_100m", "temperature_2m"} <= set(out.columns) + assert {"wind_direction_100m_sin", "wind_direction_100m_cos"} <= set(out.columns) + assert "wind_speed_100m_sin" not in out.columns + # sin/cos are consistent with the raw degrees + np.testing.assert_allclose( + out["wind_direction_100m_sin"].to_numpy(), np.sin(np.deg2rad(aligned["wind_direction_100m"].to_numpy())) + ) + + +class TestOutcomeAndGuard: + def test_extract_outcome_returns_test_power_on_index(self) -> None: + idx = _index(10) + scada = _scada(idx) + y = extract_outcome(scada, test_wtg="T1", turbine_col=_TURBINE, active_power_col=_POWER) + assert y.index.equals(idx) + expected = scada[scada[_TURBINE] == "T1"][_POWER].to_numpy() + np.testing.assert_allclose(y.to_numpy(), expected) + + def test_guard_rejects_test_turbine_feature(self) -> None: + with pytest.raises(ValueError, match="reference-only rule violated"): + check_reference_only([f"{_POWER}{QUALIFIER}R1", f"{_POWER}{QUALIFIER}T1"], test_wtg="T1") + + def test_guard_passes_for_reference_only_features(self) -> None: + check_reference_only([f"{_POWER}{QUALIFIER}R1", "temperature_2m"], test_wtg="T1") diff --git a/tests/benchmarking/baselines/test_power_model_fitting.py b/tests/benchmarking/baselines/test_power_model_fitting.py new file mode 100644 index 00000000..403ff9bc --- /dev/null +++ b/tests/benchmarking/baselines/test_power_model_fitting.py @@ -0,0 +1,35 @@ +"""Unit tests for the power model's time-blocked fold assignment (the baseline holdout fit).""" + +from __future__ import annotations + +import numpy as np +import pytest + +from benchmarking.baselines.power_model.fitting import time_block_folds + + +class TestTimeBlockFolds: + def test_round_robin_contiguous_blocks(self) -> None: + folds = time_block_folds(1000, n_folds=5, n_blocks=25) + assert folds.shape == (1000,) + assert set(folds) == {0, 1, 2, 3, 4} + # 25 equal blocks of 40 rows, block i -> fold i % 5 + for i in range(25): + block = folds[i * 40 : (i + 1) * 40] + assert (block == i % 5).all() + + def test_every_fold_is_a_fifth(self) -> None: + folds = time_block_folds(10_000, n_folds=5, n_blocks=25) + counts = np.bincount(folds) + assert (counts == 2000).all() + + def test_uneven_n_still_covers_all_folds(self) -> None: + folds = time_block_folds(103, n_folds=5, n_blocks=25) + assert len(folds) == 103 + assert set(folds) == {0, 1, 2, 3, 4} + + def test_bad_args_raise(self) -> None: + with pytest.raises(ValueError, match="n_folds"): + time_block_folds(100, n_folds=1, n_blocks=25) + with pytest.raises(ValueError, match="n_folds"): + time_block_folds(100, n_folds=5, n_blocks=3) diff --git a/tests/benchmarking/baselines/test_power_model_matching.py b/tests/benchmarking/baselines/test_power_model_matching.py new file mode 100644 index 00000000..a2380756 --- /dev/null +++ b/tests/benchmarking/baselines/test_power_model_matching.py @@ -0,0 +1,158 @@ +"""Tests for the power-model coarsened-exact-matching (CEM) utility (Issue 8, Component 2). + +Pure/fast unit tests on tiny hand-computable fixtures: equal counts per retained cell, one-sided +cells dropped (the common-support guard), seeded-subsample reproducibility, finite-value handling, +the balance diagnostic, and the retain-too-little warning / hard-floor raise. +""" + +from __future__ import annotations + +import logging + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.power_model.matching import coarsened_exact_match + +# One matching variable "ws" with edges [0, 10, 20, 30] -> three cells A=(0,10] B=(10,20] C=(20,30]. +_EDGES = {"ws": [0.0, 10.0, 20.0, 30.0]} + + +def _fixture() -> tuple[pd.DataFrame, np.ndarray, np.ndarray]: + """A tiny 11-row frame with hand-computable cell membership. + + positions -> (ws, role): + A: base 0,1,2 + up 3 -> two-sided, k=1 (subsample baseline 3->1) + B: base 4,5 + up 6,7 -> two-sided, k=2 (keep all) + C: base 8,9 -> one-sided, dropped (common-support guard) + base 10 has NaN ws -> dropped by the finite-value restriction + """ + ws = [5.0, 6.0, 7.0, 5.0, 15.0, 16.0, 15.0, 16.0, 25.0, 26.0, np.nan] + frame = pd.DataFrame({"ws": ws}, index=pd.RangeIndex(len(ws))) + baseline_sel = np.zeros(len(ws), dtype=bool) + upgraded_sel = np.zeros(len(ws), dtype=bool) + baseline_sel[[0, 1, 2, 4, 5, 8, 9, 10]] = True + upgraded_sel[[3, 6, 7]] = True + return frame, baseline_sel, upgraded_sel + + +def _match(seed: int = 0): # noqa: ANN202 + frame, baseline_sel, upgraded_sel = _fixture() + return coarsened_exact_match( + frame, baseline_sel=baseline_sel, upgraded_sel=upgraded_sel, bin_edges=_EDGES, seed=seed, min_matched_rows=1 + ) + + +class TestEqualCounts: + def test_equal_matched_count_per_side(self) -> None: + result = _match() + assert len(result.baseline_positions) == len(result.upgraded_positions) == 3 + + def test_equal_counts_within_every_retained_cell(self) -> None: + result = _match() + retained = result.per_cell[result.per_cell["n_matched"] > 0] + # after matching each retained cell keeps min(before) rows per side (equalised), and > 0 + assert (retained["n_matched"] == retained[["n_baseline", "n_upgraded"]].min(axis=1)).all() + assert (retained["n_matched"] > 0).all() + # cell A keeps 1/side, cell B keeps 2/side + assert sorted(retained["n_matched"].tolist()) == [1, 2] + + +class TestCommonSupport: + def test_one_sided_cell_dropped(self) -> None: + result = _match() + # cell C (positions 8, 9) is baseline-only -> excluded from the matched set + assert 8 not in result.baseline_positions + assert 9 not in result.baseline_positions + assert result.n_cells_one_sided == 1 + assert result.n_cells_two_sided == 2 + + +class TestFiniteHandling: + def test_nan_matching_value_row_excluded(self) -> None: + result = _match() + assert 10 not in result.baseline_positions # the NaN-ws baseline row never enters matching + assert result.n_baseline_in == 7 # 8 baseline rows minus the NaN one + + +class TestSeededSubsample: + def test_same_seed_is_reproducible(self) -> None: + a, b = _match(seed=3), _match(seed=3) + assert np.array_equal(a.baseline_positions, b.baseline_positions) + assert np.array_equal(a.upgraded_positions, b.upgraded_positions) + + def test_subsampled_row_comes_from_the_cell(self) -> None: + result = _match(seed=1) + # cell A keeps exactly one of the three baseline rows {0, 1, 2} + kept_from_a = [p for p in result.baseline_positions if p in (0, 1, 2)] + assert len(kept_from_a) == 1 + # cell B keeps both of its baseline rows + assert {4, 5}.issubset(set(result.baseline_positions.tolist())) + + def test_positions_are_sorted(self) -> None: + result = _match(seed=2) + assert list(result.baseline_positions) == sorted(result.baseline_positions) + assert list(result.upgraded_positions) == sorted(result.upgraded_positions) + + +class TestBalanceDiagnostic: + def test_retained_fractions(self) -> None: + result = _match() + assert result.retained_fraction_baseline == pytest.approx(3 / 7) + assert result.retained_fraction_upgraded == pytest.approx(1.0) + + def test_effective_sample_size(self) -> None: + assert _match().n_matched_per_side == 3 + + def test_per_cell_before_counts(self) -> None: + per_cell = _match().per_cell.set_index("ws") + # per-cell "before" counts by cell code (A=0, B=1, C=2) + assert per_cell.loc[0, "n_baseline"] == 3 + assert per_cell.loc[0, "n_upgraded"] == 1 + assert per_cell.loc[1, "n_baseline"] == 2 + assert per_cell.loc[1, "n_upgraded"] == 2 + assert per_cell.loc[2, "n_baseline"] == 2 + assert per_cell.loc[2, "n_upgraded"] == 0 + + +class TestGuards: + def test_raises_below_hard_floor(self) -> None: + frame, baseline_sel, upgraded_sel = _fixture() + with pytest.raises(ValueError, match="matched"): + coarsened_exact_match( + frame, baseline_sel=baseline_sel, upgraded_sel=upgraded_sel, bin_edges=_EDGES, seed=0 + ) # default min_matched_rows=10 > the 3 available + + def test_warns_when_little_retained(self, caplog: pytest.LogCaptureFixture) -> None: + frame, baseline_sel, upgraded_sel = _fixture() + with caplog.at_level(logging.WARNING): + coarsened_exact_match( # baseline retains 3/7 ≈ 0.43, below this 0.9 warn fraction + frame, + baseline_sel=baseline_sel, + upgraded_sel=upgraded_sel, + bin_edges=_EDGES, + seed=0, + min_matched_rows=1, + warn_retained_fraction=0.9, + ) + assert any("retain" in rec.message.lower() for rec in caplog.records) + + +class TestMultipleVariables: + def test_cell_key_is_the_tuple_of_all_vars(self) -> None: + # two vars: rows share ws-cell but split on a second var -> different cells, so no match + frame = pd.DataFrame({"ws": [5.0, 5.0, 5.0, 5.0], "gust": [2.0, 2.0, 18.0, 18.0]}) + baseline_sel = np.array([True, False, True, False]) + upgraded_sel = np.array([False, True, False, True]) + result = coarsened_exact_match( + frame, + baseline_sel=baseline_sel, + upgraded_sel=upgraded_sel, + bin_edges={"ws": [0.0, 10.0], "gust": [0.0, 10.0, 20.0]}, + seed=0, + min_matched_rows=1, + ) + # (ws bin 0, gust bin 0) = {base 0, up 1} and (ws 0, gust 1) = {base 2, up 3}: both two-sided, k=1 + assert result.n_matched_per_side == 2 + assert result.n_cells_two_sided == 2 diff --git a/tests/benchmarking/baselines/test_power_model_method.py b/tests/benchmarking/baselines/test_power_model_method.py new file mode 100644 index 00000000..0d2f9c9d --- /dev/null +++ b/tests/benchmarking/baselines/test_power_model_method.py @@ -0,0 +1,761 @@ +"""Recovery / correctness tests for ``PowerModelMethod`` (the §8-analog bias guard). + +Builds a toy dataset where the test turbine's power is a known function of the references plus a +known multiplicative uplift in the upgraded window, and asserts the counterfactual power model +recovers the uplift — for both prepost and toggle. Also checks the reference-only rule end-to-end +(a leak-bait test-turbine column cannot change the estimate). +""" + +from __future__ import annotations + +from dataclasses import replace +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.power_model import CURATED_ERA5_EXCLUDE, PowerModelMethod + +if TYPE_CHECKING: + from pathlib import Path +from benchmarking.baselines.power_model.method import ( + _TIME_DECAY_CAMPAIGN_MULTIPLE, + _clip_predictions, + _combine_uplift, + _implied_shrinkage, +) +from benchmarking.harness.conditions import CONDITIONS +from benchmarking.harness.method import MethodInput +from benchmarking.harness.toggle import resolve_toggle +from benchmarking.synthetic import ColumnSchema, ToggleSchedule + +_TURBINE = "TurbineName" +_POWER = "wtc_ActPower_mean" +_AVAIL = "wtc_ScReToOp_timeon" +_WS = "wtc_AcWindSp_mean" +_WS_SD = "wtc_AcWindSp_stddev" +_POWER_MAX = "wtc_ActPower_max" +_POWER_MIN = "wtc_ActPower_min" +_POWER_SD = "wtc_ActPower_stddev" +_COLUMNS = ColumnSchema( + turbine=_TURBINE, + active_power=_POWER, + active_power_min=_POWER_MIN, + wind_speed=_WS, + wind_speed_sd=_WS_SD, + gen_rpm="wtc_GenRpm_mean", + availability=_AVAIL, +) + +# Small/fast LightGBM so the toy data (a few thousand rows) is fit well. +_FAST_PARAMS = {"n_estimators": 120, "learning_rate": 0.1, "num_leaves": 31, "min_child_samples": 20} + + +def _toy_scada(n: int, *, uplift: float, treated: np.ndarray, seed: int = 0) -> pd.DataFrame: + """Long SCADA: references drive the test power; the upgrade scales test power on ``treated`` rows. + + Weather is i.i.d. across the whole window so baseline and upgraded share a distribution (this + isolates the estimator mechanics from the prepost confounding that the real study probes). + """ + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC", name="timestamp") + rng = np.random.default_rng(seed) + r1 = rng.normal(900, 150, n) + r2 = rng.normal(850, 150, n) + r3 = rng.normal(800, 150, n) + base_test = 0.4 * r1 + 0.35 * r2 + 0.25 * r3 + rng.normal(0, 15, n) + test_power = np.where(treated, base_test * (1.0 + uplift), base_test) + frames = { + "T1": test_power, + "R1": r1, + "R2": r2, + "R3": r3, + } + parts = [ + pd.DataFrame( + { + _TURBINE: name, + _POWER: power, + _AVAIL: 600.0, + _WS: power / 100.0, + _WS_SD: power / 1000.0, + # active-power companion statistics (Issue 11 reference_stat_cols candidates) + _POWER_MAX: power * 1.15, + _POWER_MIN: power * 0.85, + _POWER_SD: np.abs(power) / 20.0, + }, + index=idx, + ) + for name, power in frames.items() + ] + return pd.concat(parts) + + +class TestRecovery: + def test_recovers_known_uplift_prepost(self) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + conditions=(), # overall-only; the conditional path needs ERA5 (not supplied here) + model_params=_FAST_PARAMS, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + out = method.estimate(mi) + assert out.p50_overall == pytest.approx(0.05, abs=0.02) + + def test_recovers_known_uplift_toggle(self) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + schedule = ToggleSchedule(period=pd.Timedelta(hours=4)) + treated = np.asarray((((idx - idx.min()) // (schedule.period / 2)).astype(int) % 2) == 1) + scada = _toy_scada(n, uplift=0.04, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + conditions=(), # overall-only; the conditional path needs ERA5 (not supplied here) + model_params=_FAST_PARAMS, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE) + out = method.estimate(mi) + assert out.p50_overall == pytest.approx(0.04, abs=0.02) + + def test_placebo_reads_near_zero(self) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.0, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + conditions=(), # overall-only; the conditional path needs ERA5 (not supplied here) + model_params=_FAST_PARAMS, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + out = method.estimate(mi) + assert out.p50_overall == pytest.approx(0.0, abs=0.02) + + +class TestConfigGuards: + def test_era5_with_missing_wind_speed_col_raises(self) -> None: + n = 200 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + treated = np.asarray(idx >= idx[n // 2]) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=replace(_COLUMNS, wind_speed="not_a_real_column"), + baseline_rated_power_kw=2300.0, + era5_hourly_df=pd.DataFrame({"wind_speed_100m": [1.0]}), + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(idx[n // 2]), turbine_col=_TURBINE) + with pytest.raises(ValueError, match="not in scada_df"): + method.estimate(mi) + + +def _prepost_case(n: int = 4000, *, uplift: float = 0.05) -> tuple[MethodInput, pd.Timestamp]: + """A toy prepost MethodInput with a known uplift, for the model-fundamentals config trials.""" + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=uplift, treated=treated) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + return mi, changeover + + +def _fundamentals_method(**overrides: object) -> PowerModelMethod: + kwargs: dict[str, object] = { + "columns": _COLUMNS, + "baseline_rated_power_kw": 2300.0, + "conditions": (), + **overrides, + } + return PowerModelMethod(**kwargs) # type: ignore[arg-type] + + +class TestModelFundamentals: + """The self-configuring time-decay weighting and the toggle campaign mask.""" + + def test_time_decay_weights_recover_uplift(self) -> None: + mi, _ = _prepost_case() + out = _fundamentals_method( + model_params=_FAST_PARAMS, adaptive_time_decay=False, time_decay_half_life_days=30.0 + ).estimate(mi) + assert out.p50_overall == pytest.approx(0.05, abs=0.02) + + def test_time_decay_weight_values(self) -> None: + # the expert fixed-half-life path (adaptive_time_decay=False) + method = _fundamentals_method(adaptive_time_decay=False, time_decay_half_life_days=10.0) + index = pd.date_range("2019-01-01", periods=5, freq="10D", tz="UTC") + # campaign interval [index[2], index[3]]: inside weighs 1, outside decays both ways + weights = method._time_decay_weights(index, campaign_start=index[2], campaign_end=index[3]) # noqa: SLF001 + np.testing.assert_allclose(weights, [0.25, 0.5, 1.0, 1.0, 0.5]) + no_decay = _fundamentals_method(adaptive_time_decay=False, time_decay_half_life_days=None) + assert no_decay._time_decay_weights(index, campaign_start=index[2], campaign_end=index[3]) is None # noqa: SLF001 + + def test_adaptive_time_decay_half_life_scales_with_campaign_duration(self) -> None: + # the self-configuring default: half_life = k * campaign_duration_days, in both modes + method = _fundamentals_method() # adaptive_time_decay defaults to True + assert method.adaptive_time_decay is True + start = pd.Timestamp("2019-04-01", tz="UTC") + for duration_days in (30.0, 90.0, 365.0): + end = start + pd.Timedelta(days=duration_days) + hl = method._effective_half_life(campaign_start=start, campaign_end=end) # noqa: SLF001 + assert hl == pytest.approx(_TIME_DECAY_CAMPAIGN_MULTIPLE * duration_days) + + def test_adaptive_time_decay_weight_values(self) -> None: + method = _fundamentals_method() # adaptive default + index = pd.date_range("2019-01-01", periods=5, freq="10D", tz="UTC") + start, end = index[2], index[3] # 10-day campaign -> half_life = k * 10 + hl = _TIME_DECAY_CAMPAIGN_MULTIPLE * 10.0 + days_outside = np.array([20.0, 10.0, 0.0, 0.0, 10.0]) # distance to [index[2], index[3]] + expected = 0.5 ** (days_outside / hl) + weights = method._time_decay_weights(index, campaign_start=start, campaign_end=end) # noqa: SLF001 + np.testing.assert_allclose(weights, expected) + + def test_effective_half_life_fixed_and_off(self) -> None: + start = pd.Timestamp("2019-04-01", tz="UTC") + end = start + pd.Timedelta(days=90) + fixed = _fundamentals_method(adaptive_time_decay=False, time_decay_half_life_days=42.0) + assert fixed._effective_half_life(campaign_start=start, campaign_end=end) == 42.0 # noqa: SLF001 + off = _fundamentals_method(adaptive_time_decay=False, time_decay_half_life_days=None) + assert off._effective_half_life(campaign_start=start, campaign_end=end) is None # noqa: SLF001 + + def test_adaptive_with_explicit_half_life_conflict_raises(self) -> None: + mi, _ = _prepost_case(n=200) + with pytest.raises(ValueError, match="adaptive_time_decay"): + _fundamentals_method(adaptive_time_decay=True, time_decay_half_life_days=90.0).estimate(mi) + + def test_time_decay_half_life_must_be_positive(self) -> None: + mi, _ = _prepost_case(n=200) + with pytest.raises(ValueError, match="must be positive"): + _fundamentals_method(adaptive_time_decay=False, time_decay_half_life_days=0.0).estimate(mi) + + def test_started_toggle_baselines_split_pre_campaign_from_off_blocks(self) -> None: + # The old ``_campaign_mask`` folded into the shared ``resolve_toggle``: the strict + # campaign_baseline (the conditional matching's off rows) excludes pre-campaign, while the + # lenient training_baseline (the headline fit's rows) includes them. period=20D, half=10D. + index = pd.date_range("2019-01-01", periods=4, freq="10D", tz="UTC") + rows = resolve_toggle(ToggleSchedule(period=pd.Timedelta(days=20), start=index[2]), index) + pre = np.asarray(index < index[2]) # index[0], index[1] + assert not rows.campaign_baseline[pre].any() # off-only baseline drops pre-campaign + assert rows.training_baseline[pre].all() # fitting baseline keeps pre-campaign + # prepost: both baselines are exactly the pre-changeover rows (no pre-campaign concept). + prepost = resolve_toggle(pd.Timestamp(index[2]), index) + np.testing.assert_array_equal(prepost.campaign_baseline, ~prepost.upgraded) + np.testing.assert_array_equal(prepost.training_baseline, ~prepost.upgraded) + + def test_toggle_all_data_with_conditional_recovers_uplift(self) -> None: + # A toggle whose headline fit trains on the pre-campaign baseline too (the adaptive default, + # no campaign-only restriction): the conditional step still matches within the campaign only. + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + start = idx[n // 2] # first half pre-campaign baseline, second half interleaved toggle + schedule = ToggleSchedule(period=pd.Timedelta(hours=4), start=start) + within = (((idx - start) // (schedule.period / 2)).astype(int) % 2) == 1 + treated = np.asarray((idx >= start) & within) + scada = _toy_scada(n, uplift=0.04, treated=treated) + method = _fundamentals_method( + model_params=_FAST_PARAMS, + conditions=CONDITIONS, + era5_hourly_df=_toy_era5(idx), + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE) + out = method.estimate(mi) + assert out.p50_overall == pytest.approx(0.04, abs=0.02) + assert out.p50_by_condition is not None + + +class TestReferenceOnly: + def test_leak_bait_test_column_does_not_change_estimate(self) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + conditions=(), # overall-only; the conditional path needs ERA5 (not supplied here) + model_params=_FAST_PARAMS, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + baseline = method.estimate(mi).p50_overall + + # Add a column that perfectly reveals the (post-treatment) test power on the test turbine. + leaked = scada.copy() + leaked["wtc_NacWdSp_mean"] = np.where(leaked[_TURBINE] == "T1", leaked[_POWER], np.nan) + mi_leak = MethodInput( + scada_df=leaked, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE + ) + with_leak = method.estimate(mi_leak).p50_overall + # the reference-only builder ignores test-turbine columns, so the estimate is unchanged + assert with_leak == pytest.approx(baseline, abs=1e-9) + + +class TestClipPredictions: + def test_out_of_range_pulled_to_bounds_in_range_untouched(self) -> None: + # lower = min(0, 0) = 0; upper = max(2300, 1000) = 2300 + pred = np.array([-50.0, 500.0, 1500.0, 2400.0]) + clipped = _clip_predictions(pred, y_train=np.array([0.0, 500.0, 1000.0]), rated_power_kw=2300.0) + assert clipped.tolist() == [0.0, 500.0, 1500.0, 2300.0] + + def test_upper_bound_is_max_of_rated_and_train(self) -> None: + # an observed outcome above rated raises the ceiling above rated_power_kw + clipped = _clip_predictions(np.array([3000.0]), y_train=np.array([0.0, 2500.0]), rated_power_kw=2300.0) + assert clipped.tolist() == [2500.0] + + def test_floors_at_zero_for_nonnegative_training_data(self) -> None: + clipped = _clip_predictions(np.array([-5.0]), y_train=np.array([10.0, 100.0]), rated_power_kw=2300.0) + assert clipped.tolist() == [0.0] + + def test_lower_bound_allows_negative_training_data(self) -> None: + # min(0, min(y_train)) never clips a genuinely-negative observation up to 0 + clipped = _clip_predictions(np.array([-100.0]), y_train=np.array([-30.0, 100.0]), rated_power_kw=2300.0) + assert clipped.tolist() == [-30.0] + + +class TestConditionsSelection: + """``conditions`` selects which axes are reported; ``()`` skips the conditional step entirely.""" + + def _run(self, **kwargs: object) -> MethodInput: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + era5_hourly_df=_toy_era5(idx), + model_params=_FAST_PARAMS, + **kwargs, # type: ignore[arg-type] + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + return method.estimate(mi) # type: ignore[return-value] + + def test_power_only_reports_power_alone(self) -> None: + out = self._run(conditions=("power",)) + assert set(out.p50_by_condition["condition"]) == {"power"} + + def test_ws_only_reports_ws_alone(self) -> None: + out = self._run(conditions=("ws",)) + assert set(out.p50_by_condition["condition"]) == {"ws"} + + def test_default_reports_all_three(self) -> None: + # back-compat: the promoted default is unchanged for every existing caller + out = self._run() + assert set(out.p50_by_condition["condition"]) == {"ws", "ti", "power"} + + def test_empty_conditions_skips_the_conditional_step(self) -> None: + out = self._run(conditions=()) + assert out.p50_by_condition is None + + def test_unknown_condition_raises(self) -> None: + with pytest.raises(ValueError, match="unknown condition"): + PowerModelMethod(columns=_COLUMNS, baseline_rated_power_kw=2300.0, conditions=("bogus",)) + + +class TestConditionalUplift: + def test_emits_conditional_uplift_by_ws_ti_and_power(self) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.05, treated=treated) # now includes _WS_SD + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + era5_hourly_df=_toy_era5(idx), # conditional uplift (default on) matches on ERA5 weather + model_params=_FAST_PARAMS, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + out = method.estimate(mi) + bc = out.p50_by_condition + assert list(bc.columns) == ["condition", "condition_bin", "p50_uplift"] + assert set(bc["condition"]) == {"ws", "ti", "power"} + # power uses the 6 fraction-of-rated bins + assert (bc["condition"] == "power").sum() == 6 + # Issue 14: imputation fills every uncovered bin, so the reported per-bin estimate is never NaN + # (a bare NaN would let abstention game the conditional score, which drops non-finite errors). + assert bc["p50_uplift"].notna().all() + + def test_conditional_csv_carries_covered_flag(self, tmp_path: Path) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + era5_hourly_df=_toy_era5(idx), + model_params=_FAST_PARAMS, + out_dir=tmp_path, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + method.estimate(mi) + files = sorted(tmp_path.rglob("*_conditional_by_bin_*.csv")) + assert files, "no conditional_by_bin CSV written" + per_bin = pd.read_csv(files[0]) + assert "covered" in per_bin.columns + # don't assert the CSV round-trip dtype (read_csv bool inference is version-dependent); the + # column's meaning is what matters — at least some bins measured in well-populated toy data. + assert per_bin["covered"].any() + assert per_bin["p50_uplift"].notna().all() # measured-or-imputed, never bare NaN + + def test_count_floor_marks_sparse_bins_uncovered(self, tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + # force every bin below an impossibly-high floor (string target avoids a function-level import) + monkeypatch.setattr("benchmarking.baselines.power_model.method._MIN_BIN_MATCHED_COUNT", 10**9) + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + era5_hourly_df=_toy_era5(idx), + model_params=_FAST_PARAMS, + out_dir=tmp_path, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + method.estimate(mi) + per_bin = pd.read_csv(sorted(tmp_path.rglob("*_conditional_by_bin_*.csv"))[0]) + assert (~per_bin["covered"]).all() # nothing clears an impossibly-high floor + assert per_bin["p50_uplift"].notna().all() # all imputed, still never NaN + + +def _toy_era5(scada_idx: pd.DatetimeIndex, *, seed: int = 0) -> pd.DataFrame: + """Hourly ERA5 covering the toy window with the three F6 matching columns, i.i.d. over the window. + + Weather is drawn independently per hour, so the baseline and upgraded periods share a distribution + and CEM finds well-populated two-sided cells. Values sit in modest ranges so the default matching + bin edges give a handful of populated cells rather than one row each. + """ + hours = pd.date_range( + scada_idx.min().floor("h") - pd.Timedelta(hours=2), scada_idx.max().ceil("h") + pd.Timedelta(hours=2), freq="h" + ) + rng = np.random.default_rng(seed + 7) + ws = rng.uniform(4.0, 12.0, len(hours)) + return pd.DataFrame( + { + "wind_speed_100m": ws, + "wind_gusts_10m": ws * 1.4 + rng.uniform(0.0, 2.0, len(hours)), + "wind_direction_100m": rng.uniform(200.0, 260.0, len(hours)), + # extra raw columns so the Issue 9 derivations have their inputs + "wind_speed_10m": ws * 0.75, + "wind_direction_10m": rng.uniform(190.0, 250.0, len(hours)), + "temperature_2m": rng.uniform(0.0, 15.0, len(hours)), + "surface_pressure": rng.uniform(980.0, 1030.0, len(hours)), + "relative_humidity_2m": rng.uniform(50.0, 100.0, len(hours)), + }, + index=hours, + ) + + +class TestFeatureConfig: + """The surviving feature config (Issue 11 reference stats, era5_exclude, availability): columns + reach the model and estimates stay sound.""" + + def _prepost_mi(self, n: int = 4000, *, uplift: float = 0.05) -> MethodInput: + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + treated = np.asarray(idx >= idx[n // 2]) + scada = _toy_scada(n, uplift=uplift, treated=treated) + return MethodInput( + scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(idx[n // 2]), turbine_col=_TURBINE + ) + + def _fitted_feature_names(self, out_dir: Path) -> set[str]: + files = sorted(out_dir.rglob("*_feature_importance_*.csv")) + assert files, f"no feature-importance CSV under {out_dir}" + return set(pd.read_csv(files[-1])["feature"]) + + def test_reference_stat_cols_reach_model_and_recovery_holds(self, tmp_path: Path) -> None: + mi = self._prepost_mi() + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + conditions=(), + model_params=_FAST_PARAMS, + reference_stat_cols=(_POWER_MAX, _POWER_MIN, _POWER_SD), + out_dir=tmp_path, + ) + out = method.estimate(mi) + assert out.p50_overall == pytest.approx(0.05, abs=0.02) + fitted = self._fitted_feature_names(tmp_path) + assert {f"{_POWER_SD} @ R1", f"{_POWER_MAX} @ R2", f"{_POWER_MIN} @ R3"} <= fitted + assert not any(name.endswith(" @ T1") for name in fitted) + + def test_era5_exclude_drops_column_and_direction_companions(self, tmp_path: Path) -> None: + mi = self._prepost_mi() + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + era5_hourly_df=_toy_era5(pd.DatetimeIndex(mi.scada_df.index)), + conditions=(), + model_params=_FAST_PARAMS, + era5_exclude=("wind_speed_10m", "wind_direction_10m"), + out_dir=tmp_path, + ) + out = method.estimate(mi) + assert out.p50_overall == pytest.approx(0.05, abs=0.02) + fitted = self._fitted_feature_names(tmp_path) + assert ( + not { + "wind_speed_10m", + "wind_direction_10m", + "wind_direction_10m_sin", + "wind_direction_10m_cos", + } + & fitted + ) + assert "wind_speed_100m" in fitted + + def test_era5_exclude_of_matching_var_raises_with_conditional_on(self) -> None: + mi = self._prepost_mi(n=300) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + era5_hourly_df=_toy_era5(pd.DatetimeIndex(mi.scada_df.index)), + era5_exclude=("wind_gusts_10m",), + ) + with pytest.raises(ValueError, match="matching_vars"): + method.estimate(mi) + + def test_availability_feature_off_removes_availability_columns(self, tmp_path: Path) -> None: + mi = self._prepost_mi() + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + conditions=(), + model_params=_FAST_PARAMS, + availability_feature=False, + out_dir=tmp_path, + ) + out = method.estimate(mi) + assert out.p50_overall == pytest.approx(0.05, abs=0.02) + fitted = self._fitted_feature_names(tmp_path) + assert not any(name.startswith(_AVAIL) for name in fitted) + assert f"{_POWER} @ R1" in fitted + + +class TestPromotedDefaults: + def test_effective_lgbm_params_include_tuned_min_child_samples(self) -> None: + m = PowerModelMethod(columns=_COLUMNS, baseline_rated_power_kw=2300.0) + assert m._make_model().get_params()["min_child_samples"] == 50 # noqa: SLF001 + + def test_explicit_model_params_override_the_tuned_default(self) -> None: + m = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + model_params={"min_child_samples": 123}, + ) + assert m._make_model().get_params()["min_child_samples"] == 123 # noqa: SLF001 + + def test_availability_feature_defaults_off(self) -> None: + m = PowerModelMethod(columns=_COLUMNS, baseline_rated_power_kw=2300.0) + assert m.availability_feature is False + + def test_era5_exclude_defaults_to_curated_set(self) -> None: + m = PowerModelMethod(columns=_COLUMNS, baseline_rated_power_kw=2300.0) + assert m.era5_exclude == CURATED_ERA5_EXCLUDE + + +# The re-level is now the pinned-imputed ``relevel_conditional`` in power_model.conditional; its unit +# coverage lives in test_power_model_conditional.py (TestRelevelConditionalPinned). Kept here only: +# the direction-combine helpers, still in method.py. + + +class TestCombineDirections: + def test_recovers_uplift_and_shrinkage_from_ratios(self) -> None: + # construct the two directions from a known uplift u and shrinkage s: + # 1 + r_fwd = (1 + u) / s ; 1 + r_rev = 1 / (s (1 + u)) + u, s = 0.06, 0.85 + r_fwd = (1 + u) / s - 1 + r_rev = 1 / (s * (1 + u)) - 1 + assert _combine_uplift(np.array([r_fwd]), np.array([r_rev]))[0] == pytest.approx(u) + assert _implied_shrinkage(np.array([r_fwd]), np.array([r_rev]))[0] == pytest.approx(s) + + def test_nonpositive_ratio_gives_nan(self) -> None: + # (1 + r) <= 0 on either side is unphysical -> NaN, not a complex/blown-up number + out = _combine_uplift(np.array([-1.5, 0.1]), np.array([0.1, -2.0])) + assert np.isnan(out).tolist() == [True, True] + + +class TestConditional: + def test_requires_era5(self) -> None: + n = 300 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + treated = np.asarray(idx >= idx[n // 2]) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + model_params=_FAST_PARAMS, # conditional on by default, but no era5_hourly_df -> must raise + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(idx[n // 2]), turbine_col=_TURBINE) + with pytest.raises(ValueError, match="ERA5"): + method.estimate(mi) + + def test_recovers_known_uplift_through_two_directions(self) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + era5_hourly_df=_toy_era5(idx), + model_params=_FAST_PARAMS, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + out = method.estimate(mi) + # matched i.i.d. weather -> shrinkage ~1, forward-only overall recovers the true uplift + assert out.p50_overall == pytest.approx(0.05, abs=0.02) + assert set(out.p50_by_condition["condition"]) == {"ws", "ti", "power"} + assert list(out.p50_by_condition.columns) == ["condition", "condition_bin", "p50_uplift"] + + def test_overall_matches_conditional_off_and_bins_aggregate_to_it(self, tmp_path: Path) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.03, treated=treated) + config = { + "columns": _COLUMNS, + "baseline_rated_power_kw": 2300.0, + "era5_hourly_df": _toy_era5(idx), # same features both ways, so the headline is comparable + "model_params": _FAST_PARAMS, + } + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + overall_only = PowerModelMethod(**config, conditions=()).estimate(mi).p50_overall + method = PowerModelMethod(**config, out_dir=tmp_path) # conditional on by default + out = method.estimate(mi) + + # 1. the headline is the single full-data fit; computing the conditional step leaves it unchanged + assert out.p50_overall == pytest.approx(overall_only, rel=1e-9) + # 2. self-consistency: each of the ws and ti decompositions energy-aggregates back to that overall + run_dir = next(p for p in tmp_path.iterdir() if p.is_dir()) + by_bin = pd.read_csv(next((run_dir / "conditional").glob("*_conditional_by_bin_*.csv"))) + for _cond, g in by_bin.groupby("condition"): + good = g[np.isfinite(g["p50_uplift"])] + agg = good["sum_actual"].sum() / (good["sum_actual"] / (1.0 + good["p50_uplift"])).sum() + assert agg == pytest.approx(1.0 + out.p50_overall, rel=1e-6) + + def test_writes_shrinkage_and_cem_balance_diagnostics(self, tmp_path: Path) -> None: + n = 4000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _toy_scada(n, uplift=0.05, treated=treated) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=2300.0, + era5_hourly_df=_toy_era5(idx), + out_dir=tmp_path, + model_params=_FAST_PARAMS, + ) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + method.estimate(mi) + + run_dirs = [p for p in tmp_path.iterdir() if p.is_dir()] + assert len(run_dirs) == 1 + conditional_dir = run_dirs[0] / "conditional" + overall = pd.read_csv(next(conditional_dir.glob("*_conditional_overall_*.csv"))) + by_bin = pd.read_csv(next(conditional_dir.glob("*_conditional_by_bin_*.csv"))) + balance = pd.read_csv(next(conditional_dir.glob("*_cem_balance_*.csv"))) + assert next(conditional_dir.glob("*_cem_cells_*.csv"), None) is not None + # implied shrinkage s is surfaced overall and per-bin; matched weather -> s ~ 1 + assert "implied_shrinkage" in overall.columns + assert overall["implied_shrinkage"].iloc[0] == pytest.approx(1.0, abs=0.1) + assert {"condition", "condition_bin", "r_fwd", "r_rev", "implied_shrinkage", "p50_uplift"} <= set( + by_bin.columns + ) + # CEM balance carries the coverage numbers + assert {"n_matched_per_side", "retained_fraction_baseline", "n_cells_one_sided"} <= set(balance.columns) + + +def _shrinkage_scada(n: int, *, uplift: float, treated: np.ndarray, seed: int = 0) -> pd.DataFrame: + """Attenuation-shrinkage toy: references are *noisy* proxies of a steep power curve. + + Because the references (the model's features) are noisy measurements of the same weather-driven + power, the counterfactual model learns an attenuated conditional mean — it over-predicts where power + is low and under-predicts where it is high (multiplicative shrinkage). The test wind speed is the + *clean* driver, so binning by it exposes that compression as a spurious per-bin uplift tilt even at + the placebo (the F5 mechanism). Weather is i.i.d. across the window, so baseline and upgraded are + distribution-matched and the shrinkage is common to both cross-predict directions -> it cancels. + """ + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC", name="timestamp") + rng = np.random.default_rng(seed) + w = rng.uniform(3.0, 12.0, n) # latent wind speed, i.i.d. -> matched across periods + curve = 20.0 * w**2 # steep power curve (≈180..2880 kW), so per-ws-bin compression is visible + test_power = np.where(treated, curve * (1.0 + uplift), curve) + rng.normal(0.0, 20.0, n) + parts = [ + pd.DataFrame( + { + _TURBINE: "T1", + _POWER: test_power, + _POWER_MIN: test_power * 0.85, + _AVAIL: 600.0, + _WS: w, + _WS_SD: 0.05 * w, + }, + index=idx, + ) + ] + for i in range(1, 4): + ref_power = curve + rng.normal(0.0, 500.0, n) # noisy proxy of the curve -> attenuation shrinkage + parts.append( + pd.DataFrame( + { + _TURBINE: f"R{i}", + _POWER: ref_power, + _POWER_MIN: ref_power * 0.85, + _AVAIL: 600.0, + _WS: w, + _WS_SD: 0.05 * w, + }, + index=idx, + ) + ) + return pd.concat(parts) + + +def _ws_bin_bias(by_condition: pd.DataFrame) -> pd.Series: + """Per-ws-bin uplift indexed by bin (truth is 0 at placebo, so the value *is* the bias).""" + ws = by_condition[by_condition["condition"] == "ws"] + return ws.set_index("condition_bin")["p50_uplift"] + + +class TestConditionalRegression: + def test_conditional_flat_at_shrinkage_placebo(self) -> None: + # Bias guard (design note §8-analog): on a placebo whose references are noisy proxies of a steep + # power curve, a single counterfactual fit shrinks and reads a spurious per-ws-bin uplift tilt + # (the F5 mechanism). The two-direction matched conditional cancels that common shrinkage, so the + # (default) conditional uplift must read ~flat-zero in every bin against the flat-0 truth. + n = 5000 + idx = pd.date_range("2019-01-01", periods=n, freq="10min", tz="UTC") + changeover = idx[n // 2] + treated = np.asarray(idx >= changeover) + scada = _shrinkage_scada(n, uplift=0.0, treated=treated) # placebo: true uplift 0 in every bin + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=pd.Timestamp(changeover), turbine_col=_TURBINE) + method = PowerModelMethod( + columns=_COLUMNS, + baseline_rated_power_kw=6000.0, + era5_hourly_df=_toy_era5(idx), + model_params=_FAST_PARAMS, + ) + on_ws = _ws_bin_bias(method.estimate(mi).p50_by_condition) + bins = on_ws.dropna().index + on_bias = on_ws.loc[bins].abs().mean() + # Deterministic (fixed seeds); observed on this data: mean|bias| ≈ 0.0095, max|bias| ≈ 0.020. + # Thresholds sit ~2.5x above so a version/platform bump won't flake, but a regression in the + # matched cancellation (which would let the shrinkage tilt back in) will trip them. + assert on_bias < 0.025 + assert on_ws.loc[bins].abs().max() < 0.05 diff --git a/tests/benchmarking/baselines/test_rlearner_bias_guard.py b/tests/benchmarking/baselines/test_rlearner_bias_guard.py new file mode 100644 index 00000000..d3a241cc --- /dev/null +++ b/tests/benchmarking/baselines/test_rlearner_bias_guard.py @@ -0,0 +1,126 @@ +"""The design-note §8 bias-guard regression test. + +The upgrade distorts the test turbine's own nacelle wind speed (a post-treatment variable, +design note §3). This test proves the upgrade-invariant reference-only feature rule removes the +resulting bias, and guards against anyone re-adding a test-turbine signal to the feature set: + +* a model that (wrongly) conditions on the corrupted test wind speed reports a materially biased + effect (the post-treatment signal both leaks the treatment into the covariate and destroys + propensity overlap); +* the reference-only R-learner recovers the known uplift despite the same corrupted signal; +* the feature builder's guard rejects a test-turbine-qualified feature outright. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.rlearner.features import QUALIFIER, build_reference_features, check_upgrade_invariant +from benchmarking.baselines.rlearner.method import RLearnerMethod +from benchmarking.baselines.rlearner.rlearner import cross_fit_rlearner +from benchmarking.harness.method import MethodInput + +_TURBINE = "TurbineName" +_POWER = "wtc_ActPower_mean" +_WS = "wtc_AcWindSp_mean" +_AVAIL = "wtc_ScReToOp_timeon" +_FULLY_AVAILABLE_SECS = 3600.0 +_SMALL = {"n_estimators": 120, "num_leaves": 15, "min_child_samples": 20, "verbose": -1} +_UPLIFT = 0.05 + + +def _index(n: int) -> pd.DatetimeIndex: + return pd.date_range("2020-01-01", periods=n, freq="10min", tz="UTC", name="timestamp") + + +def _corrupted_scada(idx: pd.DatetimeIndex, treated: np.ndarray) -> pd.DataFrame: + """SCADA with a known uplift AND a test wind speed corrupted by the upgrade (post-treatment). + + The test turbine's measured wind speed is shifted hard whenever it is upgraded, so the signal + effectively encodes the treatment — the textbook post-treatment trap. + """ + rng = np.random.default_rng(0) + w = rng.uniform(4.0, 12.0, len(idx)) # true free-stream wind + frames = [] + for name in ("T1", "R1", "R2"): + ws = w + rng.normal(0, 0.2, len(idx)) + power = 80.0 * w + rng.normal(0, 5.0, len(idx)) + if name == "T1": + power = np.where(treated, power * (1.0 + _UPLIFT), power) + ws = ws - 100.0 * treated # upgrade corrupts the test anemometer (post-treatment) + frames.append(pd.DataFrame({_TURBINE: name, _POWER: power, _WS: ws, _AVAIL: _FULLY_AVAILABLE_SECS}, index=idx)) + return pd.concat(frames) + + +def _headline(x: pd.DataFrame, *, y: np.ndarray, t: np.ndarray) -> float: + fit = cross_fit_rlearner(x, y=y, t=t, n_folds=4, seed=0, **_factories()) + up = t.astype(bool) + return float(np.sum(fit.tau[up]) / np.sum(fit.mu0[up])) + + +def _factories() -> dict: + from benchmarking.baselines.rlearner.nuisance import ( # noqa: PLC0415 + make_effect_model, + make_outcome_model, + make_propensity_model, + ) + + return { + "make_outcome": lambda: make_outcome_model(**_SMALL), + "make_propensity": lambda: make_propensity_model(**_SMALL), + "make_effect": lambda: make_effect_model(**_SMALL), + } + + +def test_conditioning_on_test_ws_biases_estimate() -> None: + # Demonstration: leaking the corrupted test wind speed gives a materially wrong uplift, + # while the reference-only feature set recovers the true 5%. + idx = _index(3000) + upgrade = idx[1500] + treated = np.asarray(idx >= upgrade) + scada = _corrupted_scada(idx, treated) + + x_ref = build_reference_features(scada, test_wtg="T1", turbine_col=_TURBINE) + y = scada.loc[scada[_TURBINE] == "T1", _POWER].to_numpy(dtype=float) + t = treated.astype(float) + + x_leaky = x_ref.copy() + x_leaky["LEAKED_test_ws"] = scada.loc[scada[_TURBINE] == "T1", _WS].to_numpy(dtype=float) + + ref_only = _headline(x_ref, y=y, t=t) + leaky = _headline(x_leaky, y=y, t=t) + + assert ref_only == pytest.approx(_UPLIFT, abs=0.015) # reference-only is correct + assert abs(leaky - _UPLIFT) > 0.03 # leaking the post-treatment signal gives a materially wrong answer + + +def test_method_recovers_uplift_despite_corrupted_test_ws(tmp_path) -> None: # noqa: ANN001 + # The full method never reads the test turbine's signals, so the corruption is harmless. + idx = _index(3000) + upgrade = idx[1500] + treated = np.asarray(idx >= upgrade) + scada = _corrupted_scada(idx, treated) + out = RLearnerMethod( + active_power_col=_POWER, + wind_speed_col=_WS, + availability_col=_AVAIL, + out_dir=tmp_path, + n_folds=4, + model_params=_SMALL, + ).estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE)) + assert out.p50_overall == pytest.approx(_UPLIFT, abs=0.015) + + +def test_feature_builder_never_includes_test_turbine() -> None: + idx = _index(200) + treated = np.asarray(idx >= idx[100]) + scada = _corrupted_scada(idx, treated) + x = build_reference_features(scada, test_wtg="T1", turbine_col=_TURBINE) + assert not any(c.endswith(f"{QUALIFIER}T1") for c in x.columns) + + +def test_guard_rejects_a_test_turbine_feature() -> None: + with pytest.raises(ValueError, match="upgrade-invariant"): + check_upgrade_invariant([f"{_WS}{QUALIFIER}T1"], test_wtg="T1") diff --git a/tests/benchmarking/baselines/test_rlearner_core.py b/tests/benchmarking/baselines/test_rlearner_core.py new file mode 100644 index 00000000..f19717b6 --- /dev/null +++ b/tests/benchmarking/baselines/test_rlearner_core.py @@ -0,0 +1,132 @@ +"""Tests for the cross-fit R-learner core and its LightGBM nuisance factories. + +The core is pure (no I/O): given a feature matrix ``X``, outcome ``y`` and upgrade flag ``t`` +it returns per-row ``tau``, ``m_hat``, ``e_hat`` and ``mu0`` plus the fitted outcome/effect +models. These tests use generous synthetic data so the recovered effect is unambiguous, +covering the toggle-like (flat propensity) and confounded (before/after-like) regimes. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.rlearner.nuisance import ( + make_effect_model, + make_outcome_model, + make_propensity_model, +) +from benchmarking.baselines.rlearner.rlearner import RLearnerFit, cross_fit_rlearner + + +def _small_models() -> dict: + """Fast, quiet LightGBM factories for tests (few trees, no logging).""" + params = {"n_estimators": 80, "num_leaves": 15, "min_child_samples": 20, "verbose": -1} + return { + "make_outcome": lambda: make_outcome_model(**params), + "make_propensity": lambda: make_propensity_model(**params), + "make_effect": lambda: make_effect_model(**params), + } + + +class TestNuisanceFactories: + def test_outcome_is_regressor_that_predicts(self) -> None: + rng = np.random.default_rng(0) + x = pd.DataFrame({"f": rng.normal(size=200)}) + y = 3.0 * x["f"] + model = make_outcome_model(n_estimators=50, verbose=-1).fit(x, y) + assert model.predict(x).shape == (200,) + + def test_propensity_predicts_probability(self) -> None: + rng = np.random.default_rng(1) + x = pd.DataFrame({"f": rng.normal(size=200)}) + t = (x["f"] > 0).astype(int) + model = make_propensity_model(n_estimators=50, verbose=-1).fit(x, t) + proba = model.predict_proba(x)[:, 1] + assert ((proba >= 0) & (proba <= 1)).all() + + +class TestCrossFitRLearner: + def test_recovers_constant_effect_with_flat_propensity(self) -> None: + # toggle-like: t is random (propensity ~0.5); y = mu0(x) + tau*t + small noise + rng = np.random.default_rng(0) + n = 3000 + x = rng.uniform(0, 1, size=n) + x_df = pd.DataFrame({"ref_ws": x}) + mu0 = 10.0 * x + t = rng.binomial(1, 0.5, size=n) + tau_true = 2.0 + y = mu0 + tau_true * t + rng.normal(0, 0.1, size=n) + fit = cross_fit_rlearner(x_df, y=y, t=t, n_folds=4, seed=0, **_small_models()) + assert isinstance(fit, RLearnerFit) + assert float(np.mean(fit.tau)) == pytest.approx(tau_true, abs=0.3) + assert float(np.mean(fit.e_hat)) == pytest.approx(0.5, abs=0.05) + + def test_recovers_effect_under_confounding(self) -> None: + # before/after-like: t is confounded with x but stochastic, so overlap (positivity) holds. + # The upgraded period over-samples high-wind conditions; a naive treated-minus-baseline + # difference is badly biased, but the R-learner recovers tau by partialling x out of both. + rng = np.random.default_rng(1) + n = 5000 + x = rng.uniform(0, 1, size=n) + x_df = pd.DataFrame({"ref_ws": x}) + mu0 = 10.0 * x + propensity = 0.2 + 0.6 * x # in (0.2, 0.8): confounded with x but always overlapping + t = rng.binomial(1, propensity) + tau_true = 2.0 + y = mu0 + tau_true * t + rng.normal(0, 0.1, size=n) + naive_diff = y[t == 1].mean() - y[t == 0].mean() + assert naive_diff > 3.5 # confirm the naive estimate really is badly biased high + fit = cross_fit_rlearner(x_df, y=y, t=t, n_folds=5, seed=0, **_small_models()) + assert float(np.mean(fit.tau)) == pytest.approx(tau_true, abs=0.5) + + def test_placebo_reports_zero_effect_flat_propensity(self) -> None: + # a placebo upgrade (no real effect) must report ~0 uplift, not a spurious one. + rng = np.random.default_rng(10) + n = 4000 + x = rng.uniform(0, 1, size=n) + x_df = pd.DataFrame({"ref_ws": x}) + t = rng.binomial(1, 0.5, size=n) + y = 10.0 * x + rng.normal(0, 0.1, size=n) # no tau*t term + fit = cross_fit_rlearner(x_df, y=y, t=t, n_folds=4, seed=0, **_small_models()) + tau_mean = float(np.mean(fit.tau)) + assert tau_mean == pytest.approx(0.0, abs=0.2) + # the headline aggregation (sum tau / sum mu0) is also ~0 + assert float(np.sum(fit.tau) / np.sum(fit.mu0)) == pytest.approx(0.0, abs=0.05) + + def test_placebo_reports_zero_effect_under_confounding(self) -> None: + # placebo with the upgraded period over-sampling high wind: still ~0, no covariate-shift bias. + rng = np.random.default_rng(11) + n = 5000 + x = rng.uniform(0, 1, size=n) + x_df = pd.DataFrame({"ref_ws": x}) + t = rng.binomial(1, 0.2 + 0.6 * x) + y = 10.0 * x + rng.normal(0, 0.1, size=n) # no real effect + naive_diff = y[t == 1].mean() - y[t == 0].mean() + assert naive_diff > 1.5 # naive would wrongly report a large positive "uplift" + fit = cross_fit_rlearner(x_df, y=y, t=t, n_folds=5, seed=0, **_small_models()) + assert float(np.sum(fit.tau) / np.sum(fit.mu0)) == pytest.approx(0.0, abs=0.05) + + def test_mu0_identity_holds(self) -> None: + rng = np.random.default_rng(2) + n = 1500 + x = rng.uniform(0, 1, size=n) + x_df = pd.DataFrame({"ref_ws": x}) + t = rng.binomial(1, 0.5, size=n) + y = 10.0 * x + 2.0 * t + rng.normal(0, 0.1, size=n) + fit = cross_fit_rlearner(x_df, y=y, t=t, n_folds=4, seed=0, **_small_models()) + # mu0 = m_hat - e_hat * tau by construction + assert fit.mu0 == pytest.approx(fit.m_hat - fit.e_hat * fit.tau) + + def test_handles_nan_features(self) -> None: + # LightGBM handles NaN natively; the core must not choke on NaN in X. + rng = np.random.default_rng(3) + n = 1500 + x = rng.uniform(0, 1, size=n) + x_df = pd.DataFrame({"ref_ws": x, "noisy": rng.normal(size=n)}) + x_df.loc[x_df.index[:50], "noisy"] = np.nan + t = rng.binomial(1, 0.5, size=n) + y = 10.0 * x + 2.0 * t + rng.normal(0, 0.1, size=n) + fit = cross_fit_rlearner(x_df, y=y, t=t, n_folds=4, seed=0, **_small_models()) + assert np.isfinite(fit.tau).all() diff --git a/tests/benchmarking/baselines/test_rlearner_end_to_end.py b/tests/benchmarking/baselines/test_rlearner_end_to_end.py new file mode 100644 index 00000000..284cbb84 --- /dev/null +++ b/tests/benchmarking/baselines/test_rlearner_end_to_end.py @@ -0,0 +1,104 @@ +"""Slow end-to-end tests: the R-learner on HoT-derived synthetic datasets via the harness. + +Downloads the Hill of Towie v2 SCADA (Zenodo) and ERA5 (Open-Meteo), injects a known +constant-Cp uplift, and scores ``RLearnerMethod`` and an oracle through the harness on real +data in both prepost and toggle modes. v0 is deliberately not exercised here (a real wind_up +run per campaign is far too slow). Marked ``slow`` (network + model fitting); skipped by +``-m "not slow"``. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.rlearner import RLearnerMethod +from benchmarking.harness import StudyConfig, score_study +from benchmarking.harness.example_hot_study import OracleMethod +from benchmarking.synthetic import HOT_COLUMNS, ConstantCpChange +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +_SUBSET = ["T01", "T03", "T04", "T07"] +_WTG_NUMBERS = [1, 3, 4, 7] +_FAST_MODEL = {"n_estimators": 200, "verbose": -1} + + +def _rlearner(tmp_path) -> RLearnerMethod: # noqa: ANN001 + context = build_hot_v0_context(wtg_names=_SUBSET) + return RLearnerMethod( + active_power_col=HOT_COLUMNS.active_power, + wind_speed_col=HOT_COLUMNS.wind_speed, + availability_col=HOT_COLUMNS.availability, + era5_hourly_df=context.reanalysis_datasets[0].data, + out_dir=tmp_path / "rlearner_runs", + model_params=_FAST_MODEL, + ) + + +@pytest.mark.skip(reason="Rlearner method abandoned") +@pytest.mark.slow +def test_rlearner_recovers_prepost_uplift(tmp_path) -> None: # noqa: ANN001 + scada_df, _ = load_hot_scada( + start_dt=pd.Timestamp("2016-01-01", tz="UTC"), + end_dt_excl=pd.Timestamp("2021-01-01", tz="UTC"), + wtg_numbers=_WTG_NUMBERS, + wtg_names=_SUBSET, + ) + study = StudyConfig( + mode="prepost", + turbine_subset=_SUBSET, + treatment_start_range=(pd.Timestamp("2019-01-01", tz="UTC"), pd.Timestamp("2019-01-08", tz="UTC")), + min_pre_months=24, + campaign_months=[6], + n_replicates=1, + seed=0, + ) + results = score_study( + scada_df, + profile=[ConstantCpChange(delta=0.05)], + methods=[_rlearner(tmp_path), OracleMethod(scada_df)], + study=study, + profile_name="constant_cp_prepost", + ) + oracle_row = results.loc[results["method"] == "oracle"].iloc[0] + rlearner_row = results.loc[results["method"] == "rlearner"].iloc[0] + assert abs(oracle_row["signed_error"]) < 1e-6 + assert rlearner_row["truth"] > 0 + assert np.isfinite(rlearner_row["estimate"]) + assert abs(rlearner_row["signed_error"]) < 0.03 + + +@pytest.mark.skip(reason="Rlearner method abandoned") +@pytest.mark.slow +def test_rlearner_recovers_toggle_uplift(tmp_path) -> None: # noqa: ANN001 + scada_df, _ = load_hot_scada( + start_dt=pd.Timestamp("2016-01-01", tz="UTC"), + end_dt_excl=pd.Timestamp("2018-09-01", tz="UTC"), + wtg_numbers=_WTG_NUMBERS, + wtg_names=_SUBSET, + ) + study = StudyConfig( + mode="toggle", + turbine_subset=_SUBSET, + treatment_start_range=(pd.Timestamp("2018-02-01", tz="UTC"), pd.Timestamp("2018-02-08", tz="UTC")), + min_pre_months=24, + campaign_months=[6], + toggle_period=pd.Timedelta(minutes=40), + n_replicates=1, + seed=0, + ) + results = score_study( + scada_df, + profile=[ConstantCpChange(delta=0.05)], + methods=[_rlearner(tmp_path), OracleMethod(scada_df)], + study=study, + profile_name="constant_cp_toggle", + ) + oracle_row = results.loc[results["method"] == "oracle"].iloc[0] + rlearner_row = results.loc[results["method"] == "rlearner"].iloc[0] + assert abs(oracle_row["signed_error"]) < 1e-6 + assert rlearner_row["truth"] > 0 + assert np.isfinite(rlearner_row["estimate"]) + assert abs(rlearner_row["signed_error"]) < 0.03 diff --git a/tests/benchmarking/baselines/test_rlearner_era5_sync.py b/tests/benchmarking/baselines/test_rlearner_era5_sync.py new file mode 100644 index 00000000..b35dfd09 --- /dev/null +++ b/tests/benchmarking/baselines/test_rlearner_era5_sync.py @@ -0,0 +1,105 @@ +"""Tests for the R-learner ERA5 sync helper. + +ERA5 arrives hourly; SCADA is 10-min. These cover the upsample to the analysis timebase +and the wind-speed correlation lag sweep that aligns ERA5 to the SCADA, on small +hand-built frames (no network). +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.rlearner.era5_sync import ( + ERA5_WD, + ERA5_WS, + Era5SyncResult, + find_best_lag, + sync_era5, + upsample_era5_to_timebase, +) + +_RAW_WS = "wind_speed_100m" +_RAW_WD = "wind_direction_100m" + + +def _hourly(n: int, *, start: str = "2020-01-01") -> pd.DataFrame: + idx = pd.date_range(start=start, periods=n, freq="1h", tz="UTC", name="timestamp") + return pd.DataFrame( + {_RAW_WS: np.arange(n, dtype=float) + 1.0, _RAW_WD: np.linspace(0.0, 90.0, n)}, + index=idx, + ) + + +class TestUpsample: + def test_expands_to_ten_minute_grid(self) -> None: + out = upsample_era5_to_timebase(_hourly(3), timebase=pd.Timedelta(minutes=10)) + # 3 hours -> 00:00..02:50 on a 10-min grid = 18 rows (last hour's 5 trailing slots filled) + assert len(out) == 18 + assert out.index.freq is None or len(out) == 18 + + def test_passes_through_raw_columns_and_adds_aliases(self) -> None: + out = upsample_era5_to_timebase(_hourly(2), timebase=pd.Timedelta(minutes=10)) + # raw Open-Meteo columns are preserved (no renaming) and neutral ws/wd aliases are added + assert set(out.columns) == {_RAW_WS, _RAW_WD, ERA5_WS, ERA5_WD} + assert out[ERA5_WS].equals(out[_RAW_WS]) + assert out[ERA5_WD].equals(out[_RAW_WD]) + + def test_forward_fills_within_the_hour(self) -> None: + out = upsample_era5_to_timebase(_hourly(2), timebase=pd.Timedelta(minutes=10)) + # first hour's six 10-min slots all carry the first hour's raw value (1.0) + assert out[ERA5_WS].to_numpy()[:6] == pytest.approx(1.0) + assert out[ERA5_WS].to_numpy()[6:12] == pytest.approx(2.0) + + +class TestFindBestLag: + def test_recovers_known_positive_lag(self) -> None: + idx = pd.date_range("2020-01-01", periods=300, freq="10min", tz="UTC") + rng = np.random.default_rng(0) + era5_ws = pd.Series(rng.normal(8.0, 2.0, size=len(idx)), index=idx) + # reference lags ERA5 by 3 rows: reference[t] == era5[t-3] + reference_ws = era5_ws.shift(3) + best_lag, best_corr, sweep = find_best_lag( + reference_ws=reference_ws, era5_ws=era5_ws, timebase=pd.Timedelta(minutes=10) + ) + assert best_lag == 3 + assert best_corr == pytest.approx(1.0, abs=1e-6) + assert {"shift_rows", "corr"} <= set(sweep.columns) + + def test_zero_lag_when_aligned(self) -> None: + idx = pd.date_range("2020-01-01", periods=300, freq="10min", tz="UTC") + rng = np.random.default_rng(1) + era5_ws = pd.Series(rng.normal(8.0, 2.0, size=len(idx)), index=idx) + best_lag, _, _ = find_best_lag(reference_ws=era5_ws.copy(), era5_ws=era5_ws, timebase=pd.Timedelta(minutes=10)) + assert best_lag == 0 + + +class TestSyncEra5: + def test_returns_aligned_frame_on_target_index(self) -> None: + target = pd.date_range("2020-01-01 00:00", periods=120, freq="10min", tz="UTC") + # 24 hours of ERA5 covering the target window + era5 = _hourly(24) + rng = np.random.default_rng(2) + reference_ws = pd.Series(rng.normal(8.0, 2.0, size=len(target)), index=target) + result = sync_era5(era5, target_index=target, reference_ws=reference_ws) + assert isinstance(result, Era5SyncResult) + assert {_RAW_WS, _RAW_WD, ERA5_WS, ERA5_WD} <= set(result.aligned.columns) + assert result.aligned.index.equals(target) + + def test_applies_recovered_lag_to_columns(self) -> None: + target = pd.date_range("2020-01-01 00:00", periods=144, freq="10min", tz="UTC") + # random (non-monotonic) hourly ws so the lag is identifiable + hourly_idx = pd.date_range("2020-01-01", periods=36, freq="1h", tz="UTC", name="timestamp") + rng = np.random.default_rng(3) + era5 = pd.DataFrame( + {_RAW_WS: rng.normal(8.0, 2.0, size=36), _RAW_WD: rng.uniform(0.0, 360.0, size=36)}, + index=hourly_idx, + ) + up = upsample_era5_to_timebase(era5, timebase=pd.Timedelta(minutes=10)).reindex(target) + # reference lags the upsampled ERA5 ws by 2 rows -> sync should shift ERA5 forward by 2 + reference_ws = up[ERA5_WS].shift(2) + result = sync_era5(era5, target_index=target, reference_ws=reference_ws) + assert result.best_lag_rows == 2 + expected = up[ERA5_WS].shift(2) + pd.testing.assert_series_equal(result.aligned[ERA5_WS], expected, check_names=False) diff --git a/tests/benchmarking/baselines/test_rlearner_features.py b/tests/benchmarking/baselines/test_rlearner_features.py new file mode 100644 index 00000000..9c40b092 --- /dev/null +++ b/tests/benchmarking/baselines/test_rlearner_features.py @@ -0,0 +1,129 @@ +"""Tests for the R-learner upgrade-invariant feature builder. + +The builder turns long source-native SCADA into a wide feature matrix from reference +turbines only (never the test turbine's own signals), keeping original tag names, plus the +outcome/treatment extraction, the ERA5 sin/cos transform, the enforcement guard, and the +(currently no-op) feature-engineering seam. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.rlearner.era5_sync import ERA5_WD, ERA5_WS +from benchmarking.baselines.rlearner.features import ( + QUALIFIER, + build_reference_features, + check_upgrade_invariant, + engineered_reference_features, + era5_features, + extract_outcome_and_treatment, +) +from benchmarking.synthetic import ToggleSchedule, treated_mask + +_TURBINE = "TurbineName" +_POWER = "wtc_ActPower_mean" +_WS = "wtc_AcWindSp_mean" + + +def _index(n: int) -> pd.DatetimeIndex: + return pd.date_range("2020-01-01", periods=n, freq="10min", tz="UTC", name="timestamp") + + +def _scada(idx: pd.DatetimeIndex) -> pd.DataFrame: + """Long SCADA with test T1 and references R1, R2, each carrying power + wind speed.""" + rng = np.random.default_rng(0) + frames = [ + pd.DataFrame( + {_TURBINE: name, _POWER: rng.normal(800, 100, len(idx)), _WS: rng.normal(8, 2, len(idx))}, + index=idx, + ) + for name in ("T1", "R1", "R2") + ] + return pd.concat(frames) + + +class TestBuildReferenceFeatures: + def test_includes_only_reference_turbines(self) -> None: + idx = _index(12) + feats = build_reference_features(_scada(idx), test_wtg="T1", turbine_col=_TURBINE) + # two references x two value columns = four feature columns; none for T1 + assert len(feats.columns) == 4 + assert not any(c.endswith(f"{QUALIFIER}T1") for c in feats.columns) + assert feats.index.equals(idx) + + def test_keeps_original_tag_names(self) -> None: + idx = _index(12) + feats = build_reference_features(_scada(idx), test_wtg="T1", turbine_col=_TURBINE) + assert f"{_POWER}{QUALIFIER}R1" in feats.columns + assert f"{_WS}{QUALIFIER}R2" in feats.columns + + def test_preserves_nan_rows(self) -> None: + idx = _index(12) + scada = _scada(idx) + scada.loc[(scada[_TURBINE] == "R1") & (scada.index == idx[3]), _WS] = np.nan + feats = build_reference_features(scada, test_wtg="T1", turbine_col=_TURBINE) + # the NaN is preserved (no complete-case dropping) and the row is kept + assert len(feats) == len(idx) + assert np.isnan(feats.loc[idx[3], f"{_WS}{QUALIFIER}R1"]) + + def test_raises_when_no_references(self) -> None: + idx = _index(12) + only_test = _scada(idx) + only_test = only_test[only_test[_TURBINE] == "T1"] + with pytest.raises(ValueError, match="reference"): + build_reference_features(only_test, test_wtg="T1", turbine_col=_TURBINE) + + +class TestGuard: + def test_rejects_test_turbine_column(self) -> None: + names = [f"{_POWER}{QUALIFIER}R1", f"{_WS}{QUALIFIER}T1"] + with pytest.raises(ValueError, match="upgrade-invariant"): + check_upgrade_invariant(names, test_wtg="T1") + + def test_passes_reference_only_columns(self) -> None: + names = [f"{_POWER}{QUALIFIER}R1", f"{_WS}{QUALIFIER}R2"] + check_upgrade_invariant(names, test_wtg="T1") # no raise + + +class TestOutcomeTreatment: + def test_prepost_outcome_and_treatment(self) -> None: + idx = _index(20) + scada = _scada(idx) + upgrade = idx[10] + y, t = extract_outcome_and_treatment( + scada, test_wtg="T1", turbine_col=_TURBINE, active_power_col=_POWER, upgrade_timing=upgrade + ) + expected_y = scada.loc[scada[_TURBINE] == "T1", _POWER] + assert y.to_numpy() == pytest.approx(expected_y.to_numpy()) + assert t.to_numpy().tolist() == [0] * 10 + [1] * 10 + + def test_toggle_treatment(self) -> None: + idx = _index(40) + scada = _scada(idx) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=idx[0]) + _, t = extract_outcome_and_treatment( + scada, test_wtg="T1", turbine_col=_TURBINE, active_power_col=_POWER, upgrade_timing=schedule + ) + assert t.to_numpy().tolist() == np.asarray(treated_mask(idx, schedule)).astype(int).tolist() + + +class TestEra5Features: + def test_wd_becomes_sin_cos(self) -> None: + idx = _index(4) + aligned = pd.DataFrame({ERA5_WS: [8.0, 9.0, 10.0, 11.0], ERA5_WD: [0.0, 90.0, 180.0, 270.0]}, index=idx) + feats = era5_features(aligned) + assert list(feats.columns) == [ERA5_WS, "era5_wd_sin", "era5_wd_cos"] + assert feats["era5_wd_sin"].to_numpy() == pytest.approx([0.0, 1.0, 0.0, -1.0], abs=1e-9) + assert feats["era5_wd_cos"].to_numpy() == pytest.approx([1.0, 0.0, -1.0, 0.0], abs=1e-9) + + +class TestEngineeredSeam: + def test_currently_adds_no_columns(self) -> None: + idx = _index(12) + eng = engineered_reference_features(_scada(idx), test_wtg="T1", turbine_col=_TURBINE) + # the seam exists (north-corrected yaw etc. go here later) but adds nothing yet + assert list(eng.columns) == [] + assert eng.index.equals(idx) diff --git a/tests/benchmarking/baselines/test_rlearner_filtering.py b/tests/benchmarking/baselines/test_rlearner_filtering.py new file mode 100644 index 00000000..26eac7f7 --- /dev/null +++ b/tests/benchmarking/baselines/test_rlearner_filtering.py @@ -0,0 +1,107 @@ +"""Tests for the R-learner test-turbine normal-operation filter. + +The outcome ``Y`` is the test turbine's power, so abnormal operation (downtime, curtailment, +frozen/stuck sensors) unrelated to the upgrade would otherwise be attributed to it. These +cover the stuck-data and downtime filters, and the central correctness rule: filter on +*cause* (operational flags) never *effect* (low power). +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd + +from benchmarking.baselines.rlearner.filtering import NormalOperationFilter + +_POWER = "wtc_ActPower_mean" +_WS = "wtc_AcWindSp_mean" +_AVAIL = "wtc_ScReToOp_timeon" # seconds ready to operate in the period +_TIMEBASE = pd.Timedelta(minutes=10) +_FULL = _TIMEBASE.total_seconds() # 600s = fully available + + +def _test_rows(n: int = 10) -> pd.DataFrame: + idx = pd.date_range("2020-01-01", periods=n, freq="10min", tz="UTC", name="timestamp") + rng = np.random.default_rng(0) + return pd.DataFrame( + { + _POWER: rng.uniform(200, 900, n), + _WS: rng.uniform(5, 12, n), + _AVAIL: np.full(n, _FULL), + }, + index=idx, + ) + + +class TestDowntime: + def test_drops_partial_availability(self) -> None: + rows = _test_rows() + rows.loc[rows.index[2], _AVAIL] = 300.0 # only half the period available + keep = NormalOperationFilter(active_power_col=_POWER, availability_col=_AVAIL).keep_mask( + rows, timebase=_TIMEBASE + ) + assert not keep.iloc[2] + assert keep.drop(rows.index[2]).all() + + def test_drops_nan_availability(self) -> None: + rows = _test_rows() + rows.loc[rows.index[4], _AVAIL] = np.nan + keep = NormalOperationFilter(active_power_col=_POWER, availability_col=_AVAIL).keep_mask( + rows, timebase=_TIMEBASE + ) + assert not keep.iloc[4] + + def test_no_availability_col_keeps_all(self) -> None: + rows = _test_rows() + keep = NormalOperationFilter(active_power_col=_POWER, apply_stuck_filter=False).keep_mask( + rows, timebase=_TIMEBASE + ) + assert keep.all() + + +class TestStuckData: + def test_flags_frozen_rows_in_normal_wind(self) -> None: + rows = _test_rows() + # rows 5,6,7 frozen (every signal identical to row 4) in normal wind -> stuck + for i in (5, 6, 7): + rows.iloc[i] = rows.iloc[4] + keep = NormalOperationFilter(active_power_col=_POWER, wind_speed_col=_WS).keep_mask(rows, timebase=_TIMEBASE) + assert not keep.iloc[[5, 6, 7]].any() + assert keep.iloc[4] # the first of the run is not itself a repeat + + def test_low_wind_calm_is_not_stuck(self) -> None: + rows = _test_rows() + # frozen run but at very low wind: a genuine calm, must NOT be filtered as stuck + for i in (5, 6, 7): + rows.iloc[i] = rows.iloc[4] + rows.iloc[i, rows.columns.get_loc(_WS)] = 0.5 + rows.iloc[4, rows.columns.get_loc(_WS)] = 0.5 + keep = NormalOperationFilter(active_power_col=_POWER, wind_speed_col=_WS).keep_mask(rows, timebase=_TIMEBASE) + assert keep.iloc[[5, 6, 7]].all() + + +class TestCauseNotEffect: + def test_low_power_but_normal_operation_is_kept(self) -> None: + # the key rule: a genuinely low-power record that is fully available and not stuck must be + # KEPT. Filtering it (because power is low) would remove real low-uplift records and bias. + rows = _test_rows() + rows.loc[rows.index[3], _POWER] = 5.0 # very low power, but available and not frozen + keep = NormalOperationFilter(active_power_col=_POWER, wind_speed_col=_WS, availability_col=_AVAIL).keep_mask( + rows, timebase=_TIMEBASE + ) + assert keep.iloc[3] + + def test_nan_power_is_dropped(self) -> None: + rows = _test_rows() + rows.loc[rows.index[6], _POWER] = np.nan + keep = NormalOperationFilter(active_power_col=_POWER).keep_mask(rows, timebase=_TIMEBASE) + assert not keep.iloc[6] + + +class TestReturnType: + def test_keep_mask_is_boolean_series_on_index(self) -> None: + rows = _test_rows() + keep = NormalOperationFilter(active_power_col=_POWER).keep_mask(rows, timebase=_TIMEBASE) + assert isinstance(keep, pd.Series) + assert keep.index.equals(rows.index) + assert keep.dtype == bool diff --git a/tests/benchmarking/baselines/test_rlearner_method.py b/tests/benchmarking/baselines/test_rlearner_method.py new file mode 100644 index 00000000..02095776 --- /dev/null +++ b/tests/benchmarking/baselines/test_rlearner_method.py @@ -0,0 +1,189 @@ +"""End-to-end tests for RLearnerMethod behind the harness Method seam. + +Small, weather-driven SCADA where the reference turbines predict the test turbine's power, so +the R-learner can recover an injected uplift. Covers prepost and toggle recovery, the headline +aggregation, the written diagnostics, and the ERA5-free path. +""" + +from __future__ import annotations + +from pathlib import Path + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.rlearner.method import RLearnerMethod +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.synthetic import ToggleSchedule, treated_mask + +_TURBINE = "TurbineName" +_POWER = "wtc_ActPower_mean" +_WS = "wtc_AcWindSp_mean" +_AVAIL = "wtc_ScReToOp_timeon" +_FULLY_AVAILABLE_SECS = 3600.0 # >= any test timebase's full period, so the availability filter keeps every row +_SMALL = {"n_estimators": 120, "num_leaves": 15, "min_child_samples": 20, "verbose": -1} + + +def _index(n: int) -> pd.DatetimeIndex: + return pd.date_range("2020-01-01", periods=n, freq="10min", tz="UTC", name="timestamp") + + +def _weather_scada(idx: pd.DatetimeIndex, *, treated: np.ndarray, uplift: float) -> pd.DataFrame: + """Long SCADA: a shared wind drives every turbine; the test turbine is lifted when treated.""" + rng = np.random.default_rng(0) + w = rng.uniform(4.0, 12.0, len(idx)) # shared free-stream wind + frames = [] + for name in ("T1", "R1", "R2"): + ws = w + rng.normal(0, 0.2, len(idx)) + power = 80.0 * w + rng.normal(0, 5.0, len(idx)) + if name == "T1": + power = np.where(treated, power * (1.0 + uplift), power) + frames.append(pd.DataFrame({_TURBINE: name, _POWER: power, _WS: ws, _AVAIL: _FULLY_AVAILABLE_SECS}, index=idx)) + return pd.concat(frames) + + +def _method(tmp_path: Path) -> RLearnerMethod: + return RLearnerMethod( + active_power_col=_POWER, + wind_speed_col=_WS, + availability_col=_AVAIL, + out_dir=tmp_path, + n_folds=4, + model_params=_SMALL, + seed=0, + ) + + +class TestDowntimeFilterRequired: + """The downtime filter is mandatory: availability must be configured and present in the data.""" + + def test_availability_col_is_required(self, tmp_path: Path) -> None: + with pytest.raises(TypeError): + RLearnerMethod(active_power_col=_POWER, wind_speed_col=_WS, out_dir=tmp_path) # type: ignore[call-arg] + + def test_missing_availability_column_raises(self, tmp_path: Path) -> None: + idx = _index(200) + upgrade = idx[100] + scada = _weather_scada(idx, treated=np.asarray(idx >= upgrade), uplift=0.03).drop(columns=[_AVAIL]) + with pytest.raises(ValueError, match="availability"): + _method(tmp_path).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE) + ) + + +class TestRecovery: + def test_prepost_recovers_uplift(self, tmp_path: Path) -> None: + idx = _index(3000) + upgrade = idx[1500] + treated = np.asarray(idx >= upgrade) + scada = _weather_scada(idx, treated=treated, uplift=0.03) + out = _method(tmp_path).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE) + ) + assert isinstance(out, MethodOutput) + assert out.p50_overall == pytest.approx(0.03, abs=0.012) + assert out.p50_by_condition is None + + def test_toggle_recovers_uplift(self, tmp_path: Path) -> None: + idx = _index(3000) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=idx[0]) + treated = np.asarray(treated_mask(idx, schedule)) + scada = _weather_scada(idx, treated=treated, uplift=0.03) + out = _method(tmp_path).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE) + ) + assert out.p50_overall == pytest.approx(0.03, abs=0.012) + + def test_placebo_reports_near_zero(self, tmp_path: Path) -> None: + idx = _index(3000) + upgrade = idx[1500] + treated = np.asarray(idx >= upgrade) + scada = _weather_scada(idx, treated=treated, uplift=0.0) + out = _method(tmp_path).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE) + ) + assert out.p50_overall == pytest.approx(0.0, abs=0.01) + + +def _read_one(folder: Path, kind: str) -> pd.DataFrame: + run_dirs = [p for p in Path(folder).iterdir() if p.is_dir()] + assert len(run_dirs) == 1, f"expected one run dir, found {run_dirs}" + matches = list(run_dirs[0].glob(f"*_{kind}_*.csv")) + assert len(matches) == 1, f"expected one {kind} csv, found {matches}" + return pd.read_csv(matches[0]) + + +class TestDiagnostics: + def _run(self, tmp_path: Path, *, save_plots: bool = False) -> MethodOutput: + idx = _index(2000) + upgrade = idx[1000] + treated = np.asarray(idx >= upgrade) + scada = _weather_scada(idx, treated=treated, uplift=0.03) + method = RLearnerMethod( + active_power_col=_POWER, + wind_speed_col=_WS, + availability_col=_AVAIL, + out_dir=tmp_path, + n_folds=4, + model_params=_SMALL, + save_plots=save_plots, + ) + return method.estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE)) + + def test_writes_results_and_stats_and_importance(self, tmp_path: Path) -> None: + out = self._run(tmp_path) + results = _read_one(tmp_path, "results") + stats = _read_one(tmp_path, "data_stats") + importance = _read_one(tmp_path, "feature_importance") + assert results["uplift_frc"].iloc[0] == pytest.approx(out.p50_overall) + assert sorted(stats["segment"]) == ["all", "baseline", "upgraded"] + # feature importance names the original reference tags (no test turbine columns) + assert importance["feature"].str.contains("R1").any() + assert not importance["feature"].str.endswith("T1").any() + + def test_save_plots_writes_pngs(self, tmp_path: Path) -> None: + self._run(tmp_path, save_plots=True) + run_dir = next(p for p in Path(tmp_path).iterdir() if p.is_dir()) + pngs = {p.name for p in (run_dir / "plots").rglob("*.png")} + assert "feature_importance.png" in pngs + + +class TestToggleCampaignOnly: + """A toggle fit restricted to the campaign window has a balanced (~0.5) propensity.""" + + def _toggle_scada(self) -> tuple[pd.DataFrame, ToggleSchedule]: + idx = _index(3000) + start = idx[1500] # 1500 pre-campaign rows, then 1500 rows of interleaved on/off + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=start) + treated = np.asarray(treated_mask(idx, schedule)) + return _weather_scada(idx, treated=treated, uplift=0.03), schedule + + def test_campaign_only_balances_propensity_and_recovers(self, tmp_path: Path) -> None: + scada, schedule = self._toggle_scada() + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE) + out_on = RLearnerMethod( + active_power_col=_POWER, + wind_speed_col=_WS, + availability_col=_AVAIL, + out_dir=tmp_path / "on", + n_folds=4, + model_params=_SMALL, + ).estimate(mi) + RLearnerMethod( + active_power_col=_POWER, + wind_speed_col=_WS, + availability_col=_AVAIL, + out_dir=tmp_path / "off", + n_folds=4, + model_params=_SMALL, + toggle_campaign_only=False, + ).estimate(mi) + res_on = _read_one(tmp_path / "on", "results").iloc[0] + res_off = _read_one(tmp_path / "off", "results").iloc[0] + # campaign-only: ~50/50 on/off so propensity centres near 0.5; full-window dilutes it with + # 1500 pre-campaign baseline rows, dragging the mean propensity well below 0.5. + assert res_on["propensity_mean"] == pytest.approx(0.5, abs=0.12) + assert res_on["propensity_mean"] > res_off["propensity_mean"] + assert res_on["n_selected"] < res_off["n_selected"] + assert out_on.p50_overall == pytest.approx(0.03, abs=0.015) diff --git a/tests/benchmarking/baselines/test_study_power_model_compare.py b/tests/benchmarking/baselines/test_study_power_model_compare.py new file mode 100644 index 00000000..9d08155a --- /dev/null +++ b/tests/benchmarking/baselines/test_study_power_model_compare.py @@ -0,0 +1,688 @@ +"""Benchmark reshaping in study_power_model_compare.""" + +from __future__ import annotations + +import json +import subprocess +from typing import TYPE_CHECKING + +import pandas as pd +import pytest + +if TYPE_CHECKING: + from pathlib import Path + +from benchmarking.baselines.power_model import CURATED_ERA5_EXCLUDE, TUNED_MODEL_PARAMS +from benchmarking.baselines.study_power_model_compare import ( + _BASELINE_SCHEMA, + _MATERIAL_PP, + PREPOST_CAMPAIGN_MONTHS, + TOGGLE_CAMPAIGN_MONTHS, + _check_alignment, + _conditional_plot_subset, + _covered_longest, + _git_commit, + _load_baseline_cells, + _make_power_model, + _overlay_frame, + _resolve_results_dir, + _select_profiles, + _tally, + accept_candidate, + conditional_before_after, + conditional_before_after_table, + power_model_leaderboard, + record_baseline, +) +from benchmarking.harness import StudyConfig +from benchmarking.harness.conditions import CONDITIONS + + +def test_make_power_model_defaults_to_conditional_on(tmp_path: Path) -> None: + era5 = pd.DataFrame({"wind_speed_100m": [1.0]}) + method = _make_power_model(tmp_path, era5_hourly_df=era5) + # the compare sweep runs power_model at its default: conditional uplift on, matching on the F6 set + assert method.conditions == CONDITIONS + assert method.matching_vars == ("wind_speed_100m", "wind_gusts_10m", "wind_direction_100m") + + +def test_thinned_driver_matches_promoted_defaults(tmp_path: Path) -> None: + # the driver now passes only data-schema config; the F13/F14 behaviour lives on the class, so the + # constructed method must still carry the benchmarked defaults (Issue 14 promotion). + era5 = pd.DataFrame({"wind_speed_100m": [1.0]}) + method = _make_power_model(tmp_path, era5_hourly_df=era5) + assert method.availability_feature is False + assert method.era5_exclude == CURATED_ERA5_EXCLUDE + assert method._make_model().get_params()["min_child_samples"] == TUNED_MODEL_PARAMS["min_child_samples"] # noqa: SLF001 + # the reference active-power minimum is now carried by the schema's active_power_min role, so the + # driver needs no specialist reference_stat_cols config for it. + assert method.columns.active_power_min == "wtc_ActPower_min" + assert method.reference_stat_cols == () + + +def _overall_rows(truth: float, *, months: list[int], method: str = "power_model") -> pd.DataFrame: + """Minimal tidy overall-condition rows for the alignment guard (one case per campaign length).""" + return pd.DataFrame( + { + "method": method, + "profile": "cp_0pct", + "test_wtg": "T1", + "campaign_months": months, + "treatment_start": "2020-01-01 00:00:00+00:00", + "condition": "overall", + "truth": truth, + } + ) + + +def test_campaign_grids_are_the_extended_range() -> None: + assert PREPOST_CAMPAIGN_MONTHS == [1, 2, 3, 6, 12] + assert TOGGLE_CAMPAIGN_MONTHS == [1, 2, 3, 6, 12] + + +def test_alignment_ignores_fresh_cases_absent_from_reference() -> None: + # fresh spans 1-12 months; the frozen reference only has 3/6/12 — the extra short campaigns must + # not trip the guard, which now checks truth only on the intersection. + fresh = _overall_rows(0.05, months=[1, 2, 3, 6, 12]) + reference = _overall_rows(0.05, months=[3, 6, 12], method="v0_binned") + _check_alignment(fresh, reference) # must not raise + + +def test_alignment_still_fails_on_truth_mismatch_in_intersection() -> None: + fresh = _overall_rows(0.05, months=[1, 2, 3, 6, 12]) + reference = _overall_rows(0.09, months=[3, 6, 12], method="v0_binned") # same keys, different truth + with pytest.raises(ValueError, match="disagree on ground truth"): + _check_alignment(fresh, reference) + + +def test_power_model_leaderboard_includes_overall_and_conditional_cells() -> None: + fresh = pd.DataFrame( + { + "method": "power_model", + "profile": "ws_dependent_cp", + "campaign_months": 6, + "replicate": [0, 1, 0, 1], + "condition": ["overall", "overall", "ws", "ws"], + "condition_bin": ["overall", "overall", "(6.0, 8.0]", "(6.0, 8.0]"], + "estimate": [0.05, 0.05, 0.08, 0.06], + "truth": [0.05, 0.05, 0.07, 0.07], + "signed_error": [0.0, 0.0, 0.01, -0.01], + } + ) + lb = power_model_leaderboard(fresh) + keys = set(zip(lb["condition"], lb["condition_bin"], strict=True)) + assert ("overall", "overall") in keys + assert ("ws", "(6.0, 8.0]") in keys + assert {"profile", "campaign_months", "condition", "condition_bin", "bias", "spread", "score"} <= set(lb.columns) + + +def test_conditional_plot_subset_collapses_to_longest_campaign() -> None: + # conditional_leaderboard keeps one row per (campaign_months, condition_bin); the plot needs a + # single row per bin, so mixing campaign lengths must be collapsed (regression: a multi-campaign + # subset raised "cannot reindex on an axis with duplicate labels" in plot_conditional_uplift). + cond_lb = pd.DataFrame( + { + "method": "power_model", + "profile": "ws_dependent_cp", + "campaign_months": [3, 3, 12, 12], + "condition": "ws", + "condition_bin": ["(4.0, 6.0]", "(6.0, 8.0]", "(4.0, 6.0]", "(6.0, 8.0]"], + "mean_estimate": [0.09, 0.05, 0.10, 0.05], + "mean_truth": [0.10, 0.05, 0.10, 0.05], + "bias": [-0.01, 0.0, 0.0, 0.0], + "spread": [0.01, 0.005, 0.008, 0.004], + } + ) + subset = _conditional_plot_subset(cond_lb, "ws_dependent_cp", "ws") + assert list(subset["campaign_months"].unique()) == [12] + assert not subset["condition_bin"].duplicated().any() + + +def test_conditional_plot_subset_empty_for_absent_profile() -> None: + cond_lb = pd.DataFrame( + { + "profile": ["ws_dependent_cp"], + "condition": ["ws"], + "campaign_months": [6], + "condition_bin": ["(6.0, 8.0]"], + } + ) + assert _conditional_plot_subset(cond_lb, "other_profile", "ws").empty + + +def _fresh_cond_lb() -> pd.DataFrame: + """Fresh conditional leaderboard: covered profile, two campaigns (short must be dropped).""" + return pd.DataFrame( + { + "method": "power_model", + "profile": "ws_dependent_cp", + "condition": "ws", + "campaign_months": [3, 3, 12, 12], + "condition_bin": ["(4.0, 6.0]", "(6.0, 8.0]", "(4.0, 6.0]", "(6.0, 8.0]"], + # longest-campaign (12mo) rows are the ones the table must keep: + "mean_truth": [0.10, 0.05, 0.10, 0.05], + "mean_estimate": [0.09, 0.05, 0.105, 0.08], + "bias": [-0.01, 0.0, 0.005, 0.03], + "spread": [0.01, 0.005, 0.008, 0.004], + } + ) + + +def _baseline_cells() -> pd.DataFrame: + """Pre-change benchmark cells for the same (profile, campaign, condition, bins).""" + return pd.DataFrame( + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": ["(4.0, 6.0]", "(6.0, 8.0]"], + "bias": [0.02, 0.03], + "spread": [0.009, 0.005], + "score": [0.02, 0.03], + } + ) + + +def test_covered_longest_keeps_power_condition() -> None: + # power is a first-class conditional axis, so the before/after overlay subset must not drop it. + cond_lb = pd.DataFrame( + { + "method": "power_model", + "profile": "ws_dependent_cp", + "condition": ["ws", "ti", "power", "overall"], + "campaign_months": 12, + "condition_bin": ["(4.0, 6.0]", "(0.1, 0.15]", "(230.0, 690.0]", "overall"], + "mean_truth": [0.10, 0.05, 0.06, 0.05], + "mean_estimate": [0.10, 0.05, 0.06, 0.05], + "bias": [0.0, 0.0, 0.0, 0.0], + "spread": [0.01, 0.005, 0.004, 0.003], + } + ) + kept = _covered_longest(cond_lb) + assert set(kept["condition"]) == {"ws", "ti", "power"} # every axis kept, overall dropped + + +def test_conditional_before_after_reconstructs_est_before_and_verdict() -> None: + table = conditional_before_after_table(_fresh_cond_lb(), _baseline_cells(), material_pp=_MATERIAL_PP) + + # Only the longest campaign (12mo) survives. + assert set(table["campaign_months"].unique()) == {12} + rows = table.set_index("condition_bin") + + # est_before = (mean_truth + benchmark bias), reported in pp. + # bin (4.0, 6.0]: truth 0.10 + bias_before 0.02 = 0.12 -> 12.0 pp + assert rows.loc["(4.0, 6.0]", "est_before"] == pytest.approx(12.0) + # est_after is the fresh mean_estimate: 0.105 -> 10.5 pp + assert rows.loc["(4.0, 6.0]", "est_after"] == pytest.approx(10.5) + # |bias| moved 2.0 pp -> 0.5 pp, so d_abs_bias = -1.5 pp -> "better". + assert rows.loc["(4.0, 6.0]", "d_abs_bias"] == pytest.approx(-1.5) + assert rows.loc["(4.0, 6.0]", "verdict"] == "better" + + # bin (6.0, 8.0]: |bias| unchanged (3.0 pp -> 3.0 pp) -> neutral "~". + assert rows.loc["(6.0, 8.0]", "d_abs_bias"] == pytest.approx(0.0) + assert rows.loc["(6.0, 8.0]", "verdict"] == "~" + + +def test_conditional_before_after_verdict_band_edges() -> None: + # Three bins whose |bias| change is just inside, exactly at, and just outside the 0.1 pp band. + fresh = pd.DataFrame( + { + "method": "power_model", + "profile": "ti_dependent_cp", + "condition": "ti", + "campaign_months": 12, + "condition_bin": ["inside", "at", "outside"], + "mean_truth": [0.0, 0.0, 0.0], + # fresh |bias| in pp: 4.9, 4.9, 4.89 (via mean_estimate); benchmark |bias| = 5.0 pp below. + "mean_estimate": [0.049, 0.049, 0.0489], + "bias": [0.049, 0.049, 0.0489], + "spread": [0.0, 0.0, 0.0], + } + ) + baseline = pd.DataFrame( + { + "profile": "ti_dependent_cp", + "campaign_months": 12, + "condition": "ti", + "condition_bin": ["inside", "at", "outside"], + "bias": [0.05, 0.05, 0.05], # |bias| 5.0 pp + "spread": [0.0, 0.0, 0.0], + "score": [0.05, 0.05, 0.05], + } + ) + rows = conditional_before_after_table(fresh, baseline, material_pp=0.1).set_index("condition_bin") + # inside: d_abs_bias = 4.9 - 5.0 = -0.1 pp exactly -> not beyond a strict band -> "~". + assert rows.loc["inside", "d_abs_bias"] == pytest.approx(-0.1) + assert rows.loc["inside", "verdict"] == "~" + # outside: d_abs_bias = 4.89 - 5.0 = -0.11 pp -> beyond -0.1 -> "better". + assert rows.loc["outside", "d_abs_bias"] == pytest.approx(-0.11) + assert rows.loc["outside", "verdict"] == "better" + + +def test_conditional_before_after_keeps_only_covered_profiles() -> None: + fresh = pd.concat( + [ + _fresh_cond_lb(), + _fresh_cond_lb().assign(profile="cp_plus_10pct"), # not a covered profile + ], + ignore_index=True, + ) + baseline = pd.concat( + [ + _baseline_cells(), + _baseline_cells().assign(profile="cp_plus_10pct"), + ], + ignore_index=True, + ) + table = conditional_before_after_table(fresh, baseline, material_pp=_MATERIAL_PP) + assert set(table["profile"].unique()) == {"ws_dependent_cp"} + + +def test_overlay_frame_reconstructs_benchmark_series() -> None: + frame = _overlay_frame(_fresh_cond_lb(), _baseline_cells()) + assert set(frame["method"].unique()) == {"power_model (benchmark)", "power_model (current)"} + bench = frame[frame["method"] == "power_model (benchmark)"].set_index("condition_bin") + curr = frame[frame["method"] == "power_model (current)"].set_index("condition_bin") + # benchmark estimate reconstructed in fraction: truth 0.10 + bias_before 0.02 = 0.12. + assert bench.loc["(4.0, 6.0]", "mean_estimate"] == pytest.approx(0.12) + assert bench.loc["(4.0, 6.0]", "spread"] == pytest.approx(0.009) # benchmark's own spread + # current estimate is the fresh mean_estimate (fraction), for plot_conditional_uplift to scale. + assert curr.loc["(4.0, 6.0]", "mean_estimate"] == pytest.approx(0.105) + assert {"condition", "condition_bin", "mean_truth", "mean_estimate", "spread"} <= set(frame.columns) + + +def _write_baseline(path: Path) -> None: + """A v2 benchmark JSON with covered-profile conditional cells for the orchestrator test.""" + doc = { + "schema": _BASELINE_SCHEMA, + "modes": { + "prepost": { + "recorded_utc": "2025-01-01T00:00:00Z", + "git_commit": "abc1234", + "n_replicates": 2, + "seed": 0, + "campaign_months": [12], + "profiles": ["ws_dependent_cp"], + "cells": [ + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(4.0, 6.0]", + "bias": 0.02, + "spread": 0.009, + "score": 0.02, + }, + { + "profile": "ws_dependent_cp", + "campaign_months": 12, + "condition": "ws", + "condition_bin": "(6.0, 8.0]", + "bias": 0.03, + "spread": 0.005, + "score": 0.03, + }, + ], + } + }, + } + path.write_text(json.dumps(doc)) + + +def _fresh_results() -> pd.DataFrame: + """Tidy fresh scoring results (raw rows) for one covered profile, two ws bins, two replicates.""" + rows = [] + for rep in (0, 1): + for cbin, truth, est in (("(4.0, 6.0]", 0.10, 0.105), ("(6.0, 8.0]", 0.05, 0.08)): + rows.append( + { + "method": "power_model", + "profile": "ws_dependent_cp", + "campaign_months": 12, + "replicate": rep, + "condition": "ws", + "condition_bin": cbin, + "estimate": est, + "truth": truth, + "signed_error": est - truth, + } + ) + return pd.DataFrame(rows) + + +def test_conditional_before_after_writes_csv_and_plots(tmp_path: Path) -> None: + baseline_path = tmp_path / "baseline.json" + _write_baseline(baseline_path) + comparison_dir = tmp_path / "comparison" + comparison_dir.mkdir() + + conditional_before_after("prepost", _fresh_results(), baseline_path, comparison_dir) + + assert (comparison_dir / "conditional_benchmark_comparison_prepost.csv").exists() + assert (comparison_dir / "conditional_before_after_ws_dependent_cp_ws.png").exists() + + +def test_conditional_before_after_no_baseline_is_noop(tmp_path: Path) -> None: + comparison_dir = tmp_path / "comparison" + comparison_dir.mkdir() + # No baseline file -> warn and return without writing anything (mirrors compare_to_benchmark). + conditional_before_after("prepost", _fresh_results(), tmp_path / "missing.json", comparison_dir) + assert not list(comparison_dir.iterdir()) + + +def test_resolve_results_dir_flat_layout(tmp_path: Path) -> None: + # results_*.csv directly under the mode dir (older hand-assembled reference) resolve to that dir. + (tmp_path / "results_cp_0pct.csv").write_text("method\n") + assert _resolve_results_dir(tmp_path) == tmp_path + + +def test_resolve_results_dir_timestamped_subdir(tmp_path: Path) -> None: + # start_overnight_run nests each run under //; that single subdir must be found. + run = tmp_path / "20260708_140325" + run.mkdir() + (run / "results_cp_0pct.csv").write_text("method\n") + assert _resolve_results_dir(tmp_path) == run + + +def test_resolve_results_dir_no_results_raises(tmp_path: Path) -> None: + with pytest.raises(FileNotFoundError, match=r"no results_.*csv"): + _resolve_results_dir(tmp_path) + + +def test_resolve_results_dir_multiple_runs_is_ambiguous(tmp_path: Path) -> None: + for ts in ("20260708_140325", "20260709_090000"): + run = tmp_path / ts + run.mkdir() + (run / "results_cp_0pct.csv").write_text("method\n") + with pytest.raises(ValueError, match="multiple run subdirs"): + _resolve_results_dir(tmp_path) + + +def test_select_profiles_none_returns_all() -> None: + selected = _select_profiles(None) + assert "cp_0pct" in selected + assert len(selected) == 7 # the full overnight set + + +def test_select_profiles_restricts_and_preserves_order() -> None: + selected = _select_profiles(["cp_0pct"]) + assert list(selected) == ["cp_0pct"] + + +def test_select_profiles_rejects_unknown_name() -> None: + with pytest.raises(ValueError, match="unknown profile"): + _select_profiles(["cp_0pct", "not_a_profile"]) + + +def test_tally_uses_pp_band_in_fractional_form() -> None: + # _tally works on fractional deltas; the 0.1 pp band is 1e-3 in fraction. + threshold = _MATERIAL_PP / 100.0 + # inside the band (0.9e-3) -> neutral; outside (1.1e-3) -> worse/better. + delta = pd.Series([-0.0009, 0.0009, 0.0011, -0.0011]) + result = _tally(delta, n_cells=4, threshold=threshold) + assert result == "1 better / 1 worse (of 4)" + + +def _minimal_study() -> StudyConfig: + return StudyConfig( + mode="prepost", + turbine_subset=["WT01"], + treatment_start_range=(pd.Timestamp("2020-01-01", tz="UTC"), pd.Timestamp("2020-06-01", tz="UTC")), + min_pre_months=6, + campaign_months=[6], + n_replicates=1, + seed=0, + ) + + +def _minimal_leaderboard() -> pd.DataFrame: + return pd.DataFrame( + { + "profile": ["test_profile"], + "campaign_months": [6], + "condition": ["overall"], + "condition_bin": ["overall"], + "bias": [0.01], + "spread": [0.02], + "score": [0.03], + } + ) + + +def test_load_baseline_cells_returns_none_for_wrong_schema(tmp_path: Path) -> None: + """_load_baseline_cells must return None (not crash) when the file has an old/mismatched schema.""" + path = tmp_path / "baseline.json" + doc = { + "schema": "power_model_compare_baseline_v1", + "modes": { + "prepost": { + "recorded_utc": "2025-01-01T00:00:00Z", + "git_commit": "abc1234", + "n_replicates": 1, + "seed": 0, + "campaign_months": [6], + "profiles": ["test_profile"], + "cells": [{"profile": "test_profile", "campaign_months": 6, "bias": 0.01}], + } + }, + } + path.write_text(json.dumps(doc)) + + result = _load_baseline_cells("prepost", path) + + assert result is None + + +def test_record_baseline_drops_stale_modes_on_schema_bump(tmp_path: Path) -> None: + """When on-disk schema != _BASELINE_SCHEMA, old-schema mode cells must be dropped (not inherited).""" + path = tmp_path / "baseline.json" + # Simulate a v1 file that has a toggle entry with v1-shaped cells (no condition/condition_bin). + old_doc = { + "schema": "power_model_compare_baseline_v1", + "modes": { + "toggle": { + "recorded_utc": "2025-01-01T00:00:00Z", + "git_commit": "abc1234", + "n_replicates": 1, + "seed": 0, + "campaign_months": [6], + "profiles": ["test_profile"], + "cells": [ + {"profile": "test_profile", "campaign_months": 6, "bias": 0.01, "spread": 0.02, "score": 0.03} + ], + } + }, + } + path.write_text(json.dumps(old_doc)) + + lb = _minimal_leaderboard() + study = _minimal_study() + record_baseline({"prepost": lb}, {"prepost": study}, path) + + written = json.loads(path.read_text()) + assert written["schema"] == _BASELINE_SCHEMA + # The stale v1 toggle entry must be gone — a later toggle compare must not KeyError on missing columns. + assert "toggle" not in written.get("modes", {}), "stale v1 toggle cells must be dropped after schema bump" + # Equivalently, _load_baseline_cells for toggle returns None. + assert _load_baseline_cells("toggle", path) is None + + +def test_record_baseline_preserves_sibling_modes_when_schemas_match(tmp_path: Path) -> None: + """When on-disk schema matches _BASELINE_SCHEMA, sibling modes must be preserved (incremental update).""" + path = tmp_path / "baseline.json" + # Write a v2 file that already has a toggle entry. + existing_toggle_cells = [ + { + "profile": "test_profile", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.05, + "spread": 0.06, + "score": 0.07, + } + ] + existing_doc = { + "schema": _BASELINE_SCHEMA, + "modes": { + "toggle": { + "recorded_utc": "2025-01-01T00:00:00Z", + "git_commit": "abc1234", + "n_replicates": 1, + "seed": 0, + "campaign_months": [6], + "profiles": ["test_profile"], + "cells": existing_toggle_cells, + } + }, + } + path.write_text(json.dumps(existing_doc)) + + lb = _minimal_leaderboard() + study = _minimal_study() + record_baseline({"prepost": lb}, {"prepost": study}, path) + + written = json.loads(path.read_text()) + assert written["schema"] == _BASELINE_SCHEMA + # The sibling toggle mode must still be present (incremental update must not drop it). + assert "toggle" in written.get("modes", {}), "matching-schema sibling mode must be preserved" + assert "prepost" in written.get("modes", {}), "newly recorded prepost mode must be present" + toggle_result = _load_baseline_cells("toggle", path) + assert toggle_result is not None + + +def test_record_baseline_stamps_current_schema_over_old_file(tmp_path: Path) -> None: + """record_baseline must overwrite 'schema' with _BASELINE_SCHEMA even when an old-schema file exists.""" + path = tmp_path / "baseline.json" + # Write a v1 stub to simulate the on-disk old file. + old_doc = { + "schema": "power_model_compare_baseline_v1", + "modes": {}, + } + path.write_text(json.dumps(old_doc)) + + lb = _minimal_leaderboard() + study = _minimal_study() + record_baseline({"prepost": lb}, {"prepost": study}, path) + + written = json.loads(path.read_text()) + assert written["schema"] == _BASELINE_SCHEMA + + # The round-trip via _load_baseline_cells must now succeed (not return None). + loaded = _load_baseline_cells("prepost", path) + assert loaded is not None + cells_df, prov = loaded + assert isinstance(cells_df, pd.DataFrame) + assert prov["n_replicates"] == 1 + + +def _v2_toggle_doc() -> dict: + """A valid committed baseline (schema v2) that already holds a toggle mode.""" + return { + "schema": _BASELINE_SCHEMA, + "modes": { + "toggle": { + "recorded_utc": "2025-01-01T00:00:00Z", + "git_commit": "abc1234", + "n_replicates": 1, + "seed": 0, + "campaign_months": [6], + "profiles": ["test_profile"], + "cells": [ + { + "profile": "test_profile", + "campaign_months": 6, + "condition": "overall", + "condition_bin": "overall", + "bias": 0.05, + "spread": 0.06, + "score": 0.07, + } + ], + } + }, + } + + +def test_record_baseline_seeds_siblings_from_seed_path(tmp_path: Path) -> None: + """A candidate written to a fresh path must inherit sibling modes from the committed seed_path.""" + seed = tmp_path / "committed.json" + seed.write_text(json.dumps(_v2_toggle_doc())) + candidate = tmp_path / "candidate.json" + + record_baseline({"prepost": _minimal_leaderboard()}, {"prepost": _minimal_study()}, candidate, seed_path=seed) + + written = json.loads(candidate.read_text()) + assert set(written["modes"]) == {"prepost", "toggle"}, "candidate must carry the seeded toggle sibling" + # The committed seed itself must be untouched (only the candidate file is written). + assert set(json.loads(seed.read_text())["modes"]) == {"toggle"} + + +def test_accept_candidate_promotes_candidate_to_baseline(tmp_path: Path) -> None: + candidate = tmp_path / "candidate.json" + baseline = tmp_path / "baseline.json" + record_baseline({"prepost": _minimal_leaderboard()}, {"prepost": _minimal_study()}, candidate, git_commit="abc1234") + + accept_candidate(candidate, baseline) + + assert baseline.read_text() == candidate.read_text() + assert _load_baseline_cells("prepost", baseline) is not None + + +def test_accept_candidate_refuses_a_candidate_recorded_from_a_dirty_tree(tmp_path: Path) -> None: + """Promoting is the documented no-re-run path, so it is the likeliest way a -dirty baseline lands.""" + candidate = tmp_path / "candidate.json" + record_baseline( + {"prepost": _minimal_leaderboard()}, {"prepost": _minimal_study()}, candidate, git_commit="abc1234-dirty" + ) + + with pytest.raises(ValueError, match="recorded from a dirty tree"): + accept_candidate(candidate, tmp_path / "baseline.json") + + +def test_record_baseline_stamps_the_commit_it_was_given(tmp_path: Path) -> None: + """The sweep takes hours, so reading HEAD at write time could stamp code that never ran.""" + path = tmp_path / "baseline.json" + record_baseline({"prepost": _minimal_leaderboard()}, {"prepost": _minimal_study()}, path, git_commit="deadbee") + assert json.loads(path.read_text())["modes"]["prepost"]["git_commit"] == "deadbee" + + +def test_accept_candidate_missing_candidate_raises(tmp_path: Path) -> None: + with pytest.raises(FileNotFoundError, match="no candidate"): + accept_candidate(tmp_path / "nope.json", tmp_path / "baseline.json") + + +def test_accept_candidate_rejects_wrong_schema(tmp_path: Path) -> None: + candidate = tmp_path / "candidate.json" + candidate.write_text(json.dumps({"schema": "power_model_compare_baseline_v1", "modes": {}})) + with pytest.raises(ValueError, match="schema"): + accept_candidate(candidate, tmp_path / "baseline.json") + + +def test_git_commit_ignores_untracked_files(monkeypatch: pytest.MonkeyPatch) -> None: + """An untracked file must not read as dirty (F30). + + It cannot make a run irreproducible from its commit — `git checkout ` would not have it. + Counting it made --update-baseline impossible for anyone with a stray CLAUDE.md, and is the + likeliest reason the committed baseline is stamped `-dirty` while reproducing exactly. + """ + seen: list[list[str]] = [] + + def fake_run(cmd: list[str], **_: object) -> subprocess.CompletedProcess[str]: + seen.append(cmd) + out = "abc1234" if "rev-parse" in cmd else "" # status: clean + return subprocess.CompletedProcess(cmd, 0, stdout=out, stderr="") + + monkeypatch.setattr(subprocess, "run", fake_run) + assert _git_commit() == "abc1234" + status = next(cmd for cmd in seen if "status" in cmd) + assert "--untracked-files=no" in status, "untracked files must not count towards dirtiness" + + +def test_git_commit_marks_modified_tracked_files_dirty(monkeypatch: pytest.MonkeyPatch) -> None: + def fake_run(cmd: list[str], **_: object) -> subprocess.CompletedProcess[str]: + out = "abc1234" if "rev-parse" in cmd else " M benchmarking/baselines/power_model/method.py" + return subprocess.CompletedProcess(cmd, 0, stdout=out, stderr="") + + monkeypatch.setattr(subprocess, "run", fake_run) + assert _git_commit() == "abc1234-dirty" diff --git a/tests/benchmarking/baselines/test_study_toggle_methods_compare.py b/tests/benchmarking/baselines/test_study_toggle_methods_compare.py new file mode 100644 index 00000000..bbc486ee --- /dev/null +++ b/tests/benchmarking/baselines/test_study_toggle_methods_compare.py @@ -0,0 +1,461 @@ +"""Tests for the toggle-methods regression harness (profile selection, benchmark record/diff). + +The benchmark is split into a portable baseline plus one per platform (F30), so these also cover the +routing, the portability invariant and the missing-file paths. +""" + +from __future__ import annotations + +import json +import logging +import sys +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.study_toggle_methods_compare import ( + CAMPAIGN_WEEKS, + COMPARE_METHODS, + TOGGLE_PROFILES, + MethodReproducibility, + _load_baseline, + _portable_methods, + _reproducibility, + _select_profiles, + baseline_paths, + compare_to_benchmark, + load_merged_baseline, + methods_leaderboard, + plot_results, + record_baselines, + toggle_study, +) + +if TYPE_CHECKING: + from pathlib import Path + +_BANDS = {m: _reproducibility(m).band for m in COMPARE_METHODS} + + +def test_study_is_toggle_over_the_weeks_grid() -> None: + study = toggle_study() + assert study.mode == "toggle" + assert study.campaign_lengths == CAMPAIGN_WEEKS == [1, 2, 4, 8] + assert study.campaign_length_col == "campaign_weeks" + assert study.toggle_period is not None # toggle mode requires it + + +def test_profiles_are_a_placebo_plus_a_symmetric_pair() -> None: + # the symmetric +/-2% pair is what lets a sign error show up as an asymmetry between them + assert sorted(TOGGLE_PROFILES) == ["cp_0pct", "cp_minus_2pct", "cp_plus_2pct"] + deltas = {name: profile[0].delta for name, profile in TOGGLE_PROFILES.items()} + assert deltas == {"cp_0pct": 0.0, "cp_plus_2pct": 0.02, "cp_minus_2pct": -0.02} + + +def test_select_profiles_none_returns_all() -> None: + assert _select_profiles(None) == TOGGLE_PROFILES + + +def test_select_profiles_restricts_to_the_named_subset() -> None: + assert list(_select_profiles(["cp_0pct"])) == ["cp_0pct"] + + +def test_select_profiles_rejects_unknown_name() -> None: + with pytest.raises(ValueError, match="unknown profile"): + _select_profiles(["cp_plus_99pct"]) + + +_POWER_BINS = ["(-230.0, 230.0]", "(230.0, 690.0]"] + + +def _results(*, bias_shift: float = 0.0, methods: list[str] | None = None) -> pd.DataFrame: + """Tidy rows for both methods over the weeks grid: the headline plus per-power-bin rows. + + Truth is 0 and every estimate is ``bias_shift``, so each cell's bias is ``bias_shift`` exactly. + """ + rows = [] + cells = [("overall", "overall"), *[("power", b) for b in _POWER_BINS]] + for method in methods or COMPARE_METHODS: + for weeks in CAMPAIGN_WEEKS: + for replicate in range(2): + for condition, condition_bin in cells: + rows.append( + { + "method": method, + "profile": "cp_0pct", + "campaign_weeks": weeks, + "replicate": replicate, + "condition": condition, + "condition_bin": condition_bin, + "estimate": bias_shift, + "truth": 0.0, + "signed_error": bias_shift, + "wall_time_s": 1.0, + } + ) + return pd.DataFrame(rows) + + +def _record(lb: pd.DataFrame, directory: Path, *, git_commit: str = "abc1234") -> None: + record_baselines(lb, study=toggle_study(), git_commit=git_commit, baseline_dir=directory) + + +def test_leaderboard_records_the_headline_and_the_power_bins() -> None: + # the point of recording per-bin cells: a change can leave the headline untouched and wreck a bin, + # so the benchmark must carry both. + lb = methods_leaderboard(_results()) + per_method_cells = len(CAMPAIGN_WEEKS) * (1 + len(_POWER_BINS)) # overall + each power bin + assert len(lb) == len(COMPARE_METHODS) * per_method_cells + assert set(lb["method"]) == set(COMPARE_METHODS) + assert sorted(lb["campaign_weeks"].unique()) == CAMPAIGN_WEEKS + assert set(lb["condition"]) == {"overall", "power"} + + +def test_leaderboard_records_power_bins_for_both_methods() -> None: + lb = methods_leaderboard(_results()) + power = lb[lb["condition"] == "power"] + for method in COMPARE_METHODS: + bins = set(power[power["method"] == method]["condition_bin"]) + assert bins == set(_POWER_BINS), f"{method} must contribute per-power-bin cells" + + +def test_wall_time_is_recorded_on_the_headline_rows_only() -> None: + # wall time is per estimate, not per bin, and it is what makes "this change made the method 2x + # slower" visible. Stacking the conditional rows must not silently drop it (it did once). + lb = methods_leaderboard(_results()) + headline = lb[lb["condition"] == "overall"] + per_bin = lb[lb["condition"] == "power"] + for col in ("wall_time_s_sum", "wall_time_s_mean"): + assert col in lb.columns + assert headline[col].notna().all(), f"{col} must be recorded on the headline rows" + assert per_bin[col].isna().all(), f"{col} is meaningless per bin and must be NaN there" + + +def test_wall_time_is_not_diffed(tmp_path: Path) -> None: + # it is machine- and load-dependent, so diffing it would trip the unchanged verdict every run + lb = methods_leaderboard(_results()) + _record(lb, tmp_path) + + slower = lb.assign(wall_time_s_mean=lb["wall_time_s_mean"] * 10) + merged = compare_to_benchmark(slower, comparison_dir=tmp_path / "comparison", baseline_dir=tmp_path) + assert "d_wall_time_s_mean" not in merged.columns + for col in ("d_bias", "d_spread", "d_score"): + assert (merged[col].abs() < max(_BANDS.values())).all() + + +class TestReproducibilityRegistry: + def test_toggle_specialist_is_portable_and_power_model_is_not(self) -> None: + assert _reproducibility("toggle_specialist") == MethodReproducibility(band=1e-7, portable=True) + assert _reproducibility("power_model").portable is False + + def test_an_unclassified_method_is_assumed_machine_specific(self) -> None: + """The safe side: wrongly calling one portable is a permanent failure on the other machine.""" + assert _reproducibility("some_new_method").portable is False + + def test_portable_methods_is_derived_from_the_registry(self) -> None: + assert _portable_methods() == {"toggle_specialist"} + + +class TestBaselinePaths: + def test_routes_to_the_current_platform(self, tmp_path: Path) -> None: + portable, platform = baseline_paths(tmp_path, platform="win32") + assert portable.name.endswith("_portable.json") + assert platform.name.endswith("_win32.json") + + def test_platform_defaults_to_this_machine(self, tmp_path: Path) -> None: + _, platform = baseline_paths(tmp_path) + assert platform.name.endswith(f"_{sys.platform}.json") + + +class TestRecordSplitsByPortability: + def test_writes_a_portable_and_a_platform_file(self, tmp_path: Path) -> None: + _record(methods_leaderboard(_results()), tmp_path) + portable_path, platform_path = baseline_paths(tmp_path) + assert portable_path.exists() + assert platform_path.exists() + + def test_portable_file_holds_only_portable_methods(self, tmp_path: Path) -> None: + _record(methods_leaderboard(_results()), tmp_path) + portable_path, platform_path = baseline_paths(tmp_path) + assert json.loads(portable_path.read_text())["methods"] == ["toggle_specialist"] + assert json.loads(platform_path.read_text())["methods"] == ["power_model"] + + def test_never_writes_another_platforms_file(self, tmp_path: Path) -> None: + """The property that keeps two laptops from conflicting in git.""" + _, other = baseline_paths(tmp_path, platform="some_other_os") + _record(methods_leaderboard(_results()), tmp_path) + assert not other.exists() + + def test_records_a_machine_fingerprint(self, tmp_path: Path) -> None: + _record(methods_leaderboard(_results()), tmp_path) + _, platform_path = baseline_paths(tmp_path) + prov = json.loads(platform_path.read_text()) + assert prov["platform"] + assert prov["cpu_count"] + assert prov["python_version"] + + +class TestPortabilityInvariant: + def test_identical_portable_cells_are_not_rewritten(self, tmp_path: Path) -> None: + """No churn, no conflict; the re-record doubles as proof portability holds.""" + lb = methods_leaderboard(_results(bias_shift=0.0123456789012345)) + _record(lb, tmp_path, git_commit="aaa") + portable_path, _ = baseline_paths(tmp_path) + before = portable_path.read_bytes() + + _record(lb, tmp_path, git_commit="bbb") # later commit, same portable numbers + assert portable_path.read_bytes() == before + + def test_differing_wall_time_alone_does_not_rewrite_the_portable_file(self, tmp_path: Path) -> None: + """Wall time is recorded per cell and differs every run, but is not a number under test. + + Comparing it would rewrite the shared file on every recording — churning provenance and + handing the two laptops a conflict on the one file they both touch. + """ + lb = methods_leaderboard(_results(bias_shift=0.0123456789012345)) + _record(lb, tmp_path, git_commit="aaa") + portable_path, _ = baseline_paths(tmp_path) + before = portable_path.read_bytes() + + slower = lb.assign(wall_time_s_sum=lb["wall_time_s_sum"] * 7, wall_time_s_mean=lb["wall_time_s_mean"] * 7) + _record(slower, tmp_path, git_commit="aaa") + assert portable_path.read_bytes() == before + + def test_a_portable_difference_inside_the_band_does_not_rewrite(self, tmp_path: Path) -> None: + """Portability is a claim at the band's precision, not bit-exactness.""" + lb = methods_leaderboard(_results(bias_shift=0.01)) + _record(lb, tmp_path, git_commit="aaa") + portable_path, _ = baseline_paths(tmp_path) + before = portable_path.read_bytes() + + band = _reproducibility("toggle_specialist").band + nudged = lb.assign(bias=lb["bias"] + band / 10) + _record(nudged, tmp_path, git_commit="aaa") # inside the band: not a break, not a change + assert portable_path.read_bytes() == before + + def test_differing_portable_cells_at_the_same_commit_raise(self, tmp_path: Path) -> None: + """Same code, different machine, different numbers => portability broke.""" + _record(methods_leaderboard(_results(bias_shift=0.0)), tmp_path, git_commit="samecommit") + with pytest.raises(ValueError, match="portable baseline"): + _record(methods_leaderboard(_results(bias_shift=0.05)), tmp_path, git_commit="samecommit") + + def test_differing_portable_cells_at_a_different_commit_are_recorded(self, tmp_path: Path) -> None: + """The accepted-change path. Refusing here would block every legitimate re-record.""" + _record(methods_leaderboard(_results(bias_shift=0.0)), tmp_path, git_commit="oldcommit") + _record(methods_leaderboard(_results(bias_shift=0.05)), tmp_path, git_commit="newcommit") + + portable_path, _ = baseline_paths(tmp_path) + doc = json.loads(portable_path.read_text()) + assert doc["git_commit"] == "newcommit" + assert np.allclose([c["bias"] for c in doc["cells"]], 0.05) + + def test_machine_specific_cells_may_move_freely_at_the_same_commit(self, tmp_path: Path) -> None: + """power_model is expected to differ across machines; only portable methods are policed.""" + _record(methods_leaderboard(_results(bias_shift=0.0)), tmp_path, git_commit="samecommit") + moved_pm = pd.concat( + [ + _results(bias_shift=0.0, methods=["toggle_specialist"]), + _results(bias_shift=0.05, methods=["power_model"]), + ] + ) + _record(methods_leaderboard(moved_pm), tmp_path, git_commit="samecommit") # must not raise + + +class TestLoadMerged: + def test_merges_portable_and_platform_cells(self, tmp_path: Path) -> None: + lb = methods_leaderboard(_results()) + _record(lb, tmp_path) + loaded = load_merged_baseline(tmp_path) + assert loaded is not None + cells, prov = loaded + assert set(cells["method"]) == set(COMPARE_METHODS) + assert len(cells) == len(lb) + assert len(prov) == 2 # one provenance block per file + + def test_missing_platform_file_still_diffs_the_portable_half( + self, tmp_path: Path, caplog: pytest.LogCaptureFixture + ) -> None: + """A fresh machine gets its toggle_specialist check without recording anything.""" + _record(methods_leaderboard(_results()), tmp_path) + _, platform_path = baseline_paths(tmp_path) + platform_path.unlink() + + with caplog.at_level(logging.WARNING): + loaded = load_merged_baseline(tmp_path) + assert loaded is not None + assert set(loaded[0]["method"]) == {"toggle_specialist"} + assert "No usable benchmark" in caplog.text + + def test_missing_portable_file_still_diffs_the_platform_half(self, tmp_path: Path) -> None: + _record(methods_leaderboard(_results()), tmp_path) + portable_path, _ = baseline_paths(tmp_path) + portable_path.unlink() + + loaded = load_merged_baseline(tmp_path) + assert loaded is not None + assert set(loaded[0]["method"]) == {"power_model"} + + def test_none_when_nothing_is_recorded(self, tmp_path: Path) -> None: + assert load_merged_baseline(tmp_path) is None + + def test_a_foreign_fingerprint_on_the_platform_file_warns_but_does_not_fail( + self, tmp_path: Path, caplog: pytest.LogCaptureFixture + ) -> None: + """The hole that caused a wrong 'stale baseline' conclusion: the file could not say where it came from.""" + _record(methods_leaderboard(_results()), tmp_path) + _, platform_path = baseline_paths(tmp_path) + doc = json.loads(platform_path.read_text()) + doc["cpu_count"] = (doc["cpu_count"] or 0) + 99 + platform_path.write_text(json.dumps(doc)) + + with caplog.at_level(logging.WARNING): + loaded = load_merged_baseline(tmp_path) + assert loaded is not None # a warning, never fatal + assert "machine unlike this one" in caplog.text + + def test_a_foreign_fingerprint_on_the_portable_file_does_not_warn( + self, tmp_path: Path, caplog: pytest.LogCaptureFixture + ) -> None: + """The portable file is *meant* to come from the other laptop — that is the point of it. + + This fired on every real run and blamed a machine-specific method the portable file does not + even contain. + """ + _record(methods_leaderboard(_results()), tmp_path) + portable_path, _ = baseline_paths(tmp_path) + doc = json.loads(portable_path.read_text()) + doc["platform"] = "some_other_os" + doc["cpu_count"] = (doc["cpu_count"] or 0) + 99 + portable_path.write_text(json.dumps(doc)) + + with caplog.at_level(logging.WARNING): + loaded = load_merged_baseline(tmp_path) + assert loaded is not None + assert "machine unlike this one" not in caplog.text + + def test_null_fingerprint_fields_make_no_claim(self, tmp_path: Path, caplog: pytest.LogCaptureFixture) -> None: + """The migrated v2 file cannot recover the recording machine's cpu_count; null must not warn.""" + _record(methods_leaderboard(_results()), tmp_path) + _, platform_path = baseline_paths(tmp_path) + doc = json.loads(platform_path.read_text()) + doc["cpu_count"] = None + doc["lightgbm_version"] = None + platform_path.write_text(json.dumps(doc)) + + with caplog.at_level(logging.WARNING): + load_merged_baseline(tmp_path) + assert "recorded on a machine unlike this one" not in caplog.text + + +class TestBaselineRoundTrip: + def test_round_trips(self, tmp_path: Path) -> None: + lb = methods_leaderboard(_results()) + _record(lb, tmp_path) + portable_path, _ = baseline_paths(tmp_path) + + loaded = _load_baseline(portable_path) + assert loaded is not None + cells, provenance = loaded + assert provenance["campaign_weeks"] == CAMPAIGN_WEEKS + assert provenance["seed"] == toggle_study().seed + assert len(cells) == len(lb[lb["method"] == "toggle_specialist"]) + + def test_records_the_commit_it_was_given_not_the_current_head(self, tmp_path: Path) -> None: + # a ~15-min run can straddle a commit, so reading HEAD at write time would stamp a commit + # whose code never produced these numbers (which is exactly what happened once). + _record(methods_leaderboard(_results()), tmp_path, git_commit="deadbee") + _, platform_path = baseline_paths(tmp_path) + loaded = _load_baseline(platform_path) + assert loaded is not None + assert loaded[1]["git_commit"] == "deadbee" + + def test_load_baseline_returns_none_when_absent(self, tmp_path: Path) -> None: + assert _load_baseline(tmp_path / "nope.json") is None + + def test_load_baseline_returns_none_for_wrong_schema(self, tmp_path: Path) -> None: + path = tmp_path / "baseline.json" + path.write_text(json.dumps({"schema": "something_older", "cells": []})) + assert _load_baseline(path) is None + + +def test_unchanged_method_diffs_within_the_band(tmp_path: Path) -> None: + # the property the whole script rests on: an unchanged method reproduces its cells to numerical + # noise. Use a bias with many significant digits, so the round(8) actually bites -- a round number + # here would pass even if the rounding were broken, which is the trap the first version fell into. + lb = methods_leaderboard(_results(bias_shift=0.0123456789012345)) + _record(lb, tmp_path) + + merged = compare_to_benchmark(lb, comparison_dir=tmp_path / "comparison", baseline_dir=tmp_path) + assert not merged.empty + for col in ("d_bias", "d_spread", "d_score"): + for method, group in merged.groupby("method"): + band = _BANDS[str(method)] + assert (group[col].abs() < band).all(), f"{col} must be within {method}'s band when unchanged" + + +def test_round_trip_leaves_a_nonzero_residual_from_baseline_rounding(tmp_path: Path) -> None: + # documents *why* the band is not zero: the stored baseline is rounded, so even an identical + # re-run differs in the last digit. + lb = methods_leaderboard(_results(bias_shift=0.0123456789012345)) + _record(lb, tmp_path) + + merged = compare_to_benchmark(lb, comparison_dir=tmp_path / "comparison", baseline_dir=tmp_path) + assert (merged["d_bias"].abs() > 0).any(), "round(8) should leave a residual; if not, the band can tighten" + + +def test_moved_method_shows_a_nonzero_bias_delta(tmp_path: Path) -> None: + _record(methods_leaderboard(_results(bias_shift=0.0)), tmp_path) + + moved = methods_leaderboard(_results(bias_shift=0.01)) + merged = compare_to_benchmark(moved, comparison_dir=tmp_path / "comparison", baseline_dir=tmp_path) + assert np.allclose(merged["d_bias"].to_numpy(), 0.01) + assert (merged["d_bias"].abs() > max(_BANDS.values())).all() # 1 pp is far outside any band + + +def test_comparison_csv_is_written(tmp_path: Path) -> None: + lb = methods_leaderboard(_results()) + _record(lb, tmp_path) + comparison_dir = tmp_path / "comparison" + + compare_to_benchmark(lb, comparison_dir=comparison_dir, baseline_dir=tmp_path) + assert (comparison_dir / "benchmark_comparison.csv").exists() + + +def test_compare_without_a_baseline_is_a_noop(tmp_path: Path, caplog: pytest.LogCaptureFixture) -> None: + with caplog.at_level(logging.WARNING): + merged = compare_to_benchmark( + methods_leaderboard(_results()), comparison_dir=tmp_path / "cmp", baseline_dir=tmp_path + ) + assert merged.empty + assert "No benchmark recorded" in caplog.text + + +def test_unchanged_verdict_is_logged_per_method(tmp_path: Path, caplog: pytest.LogCaptureFixture) -> None: + lb = methods_leaderboard(_results()) + _record(lb, tmp_path) + with caplog.at_level(logging.INFO): + compare_to_benchmark(lb, comparison_dir=tmp_path / "comparison", baseline_dir=tmp_path) + for method in COMPARE_METHODS: + assert f"{method}: UNCHANGED" in caplog.text + + +def test_plots_are_written_per_profile(tmp_path: Path) -> None: + plot_results(methods_leaderboard(_results()), tmp_path) + assert (tmp_path / "campaign_curves_cp_0pct.png").exists() + + +def test_a_run_scoring_no_machine_specific_methods_does_not_destroy_the_platform_baseline(tmp_path: Path) -> None: + """Overwriting a good baseline with zero cells would silently destroy it. + + --update-baseline guards the profile set but not the method set, so a run that scored only + portable methods must leave the platform file alone rather than emptying it. + """ + _record(methods_leaderboard(_results()), tmp_path, git_commit="aaa") + _, platform_path = baseline_paths(tmp_path) + before = platform_path.read_bytes() + + portable_only = methods_leaderboard(_results(methods=["toggle_specialist"])) + _record(portable_only, tmp_path, git_commit="aaa") + assert platform_path.read_bytes() == before, "a portable-only run must not empty the platform baseline" diff --git a/tests/benchmarking/baselines/test_study_toggle_specialist_uncertainty.py b/tests/benchmarking/baselines/test_study_toggle_specialist_uncertainty.py new file mode 100644 index 00000000..cb6bf569 --- /dev/null +++ b/tests/benchmarking/baselines/test_study_toggle_specialist_uncertainty.py @@ -0,0 +1,170 @@ +"""Tests for the toggle-specialist uncertainty study driver. + +The sweep itself is far too slow to run here (64 replicates of real SCADA), so these cover the +pieces that decide whether a run's *output* is right: the block-length variants, the recovery of +block length from a variant name, and the plotting's behaviour on an arbitrary ``--block-hours`` +grid. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.study_toggle_specialist_uncertainty import ( + _block_hours_of, + _focus_block_hours, + build_methods, + calibration_tables, + independent_draws, + plot_results, +) +from benchmarking.baselines.toggle_specialist import DEFAULT_BLOCK_HOURS + +if TYPE_CHECKING: + from pathlib import Path + + +def _cases(block_hours: list[float]) -> pd.DataFrame: + """A minimal scored-cases frame with a headline and one power bin per block length.""" + rng = np.random.default_rng(0) + rows = [] + for bl in block_hours: + for replicate in range(8): + for weeks in (1, 4): + for condition, bin_label, n_up in (("overall", "overall", 5000), ("power", "(0.0, 460.0]", 40)): + err = float(rng.normal(0, 0.01)) + rows.append( + { + "method": f"toggle_specialist_bl{bl:g}", + "profile": "cp_0pct", + "replicate": replicate, + "test_wtg": "T01", + "campaign_weeks": weeks, + "condition": condition, + "condition_bin": bin_label, + "estimate": err, + "truth": 0.0, + "signed_error": err, + "sigma": 0.01, + "sigma_robust": 0.01, + "n_upgraded_records": n_up, + "n_baseline_records": n_up, + "n_blocks": 10, + "frac_resamples_finite": 1.0, + "block_hours": bl, + } + ) + return pd.DataFrame(rows) + + +class TestBuildMethods: + def test_one_variant_per_block_length_each_named_for_it(self) -> None: + methods = build_methods([6.0, 48.0]) + assert [m.name for m in methods] == ["toggle_specialist_bl6", "toggle_specialist_bl48"] + assert [m.block_hours for m in methods] == [6.0, 48.0] + + def test_variants_differ_only_in_block_length(self) -> None: + """They must produce identical uplifts; only sigma may differ.""" + a, b = build_methods([6.0, 48.0]) + assert a.conditions == b.conditions + assert a.rated_power_kw == b.rated_power_kw + assert a.n_resamples == b.n_resamples + assert a.bootstrap_seed == b.bootstrap_seed + + def test_block_hours_round_trips_through_the_variant_name(self) -> None: + for block_hours in (1.0, 6.0, 48.0, 96.0): + (method,) = build_methods([block_hours]) + assert _block_hours_of(method.name) == block_hours + + +class TestFocusBlockHours: + def test_prefers_the_method_default_when_it_was_swept(self) -> None: + assert _focus_block_hours(_cases([6.0, DEFAULT_BLOCK_HOURS, 96.0])) == DEFAULT_BLOCK_HOURS + + def test_falls_back_to_the_nearest_swept_length(self) -> None: + """``--block-hours`` is a free grid, so the default need not be in it. + + The grid is built from the default rather than hardcoded, so this keeps testing the + fallback rather than the default's current value. + """ + near, far = DEFAULT_BLOCK_HOURS * 2, DEFAULT_BLOCK_HOURS * 8 + assert _focus_block_hours(_cases([near, far])) == near + + def test_single_length_grid(self) -> None: + assert _focus_block_hours(_cases([3.0])) == 3.0 + + +class TestPlotResults: + def test_all_four_plots_are_written_for_a_grid_containing_the_default(self, tmp_path: Path) -> None: + cases = _cases([DEFAULT_BLOCK_HOURS, DEFAULT_BLOCK_HOURS * 4]) + plot_results(cases, calibration_tables(cases, n_replicates=8), tmp_path) + assert sorted(p.name for p in tmp_path.iterdir()) == [ + "coverage_by_campaign_length.png", + "coverage_vs_record_count.png", + "error_vs_sigma.png", + "sigma_vs_block_length.png", + ] + + def test_all_four_plots_are_written_when_the_default_was_not_swept(self, tmp_path: Path) -> None: + """Regression: the per-case plots hardcoded 48h, so a grid without it silently lost them. + + ``coverage_vs_record_count`` was skipped entirely and ``error_vs_sigma`` was drawn from an + empty frame (an all-NaN axis limit) — a run that looked successful but produced junk. + """ + cases = _cases([DEFAULT_BLOCK_HOURS * 16, DEFAULT_BLOCK_HOURS * 32]) + plot_results(cases, calibration_tables(cases, n_replicates=8), tmp_path) + assert sorted(p.name for p in tmp_path.iterdir()) == [ + "coverage_by_campaign_length.png", + "coverage_vs_record_count.png", + "error_vs_sigma.png", + "sigma_vs_block_length.png", + ] + + +class TestCalibrationTables: + def test_reports_every_read_keyed_by_block_length(self) -> None: + tables = calibration_tables(_cases([6.0, 48.0]), n_replicates=8) + assert set(tables) == { + "headline_by_block", + "headline_by_block_and_length", + "headline_by_block_and_profile", + "per_bin_by_block", + "per_bin_by_block_and_bin", + "per_bin_by_block_and_length", + } + assert sorted(tables["headline_by_block"]["block_hours"]) == [6.0, 48.0] + + def test_headline_and_per_bin_are_scored_separately(self) -> None: + """They fail for different reasons, so pooling them would hide both.""" + tables = calibration_tables(_cases([6.0]), n_replicates=8) + assert tables["headline_by_block"]["n"].iloc[0] == 16 # 8 replicates x 2 campaign lengths + assert tables["per_bin_by_block"]["n"].iloc[0] == 16 + + def test_the_coverage_standard_error_is_quoted_per_campaign_length(self) -> None: + """Rows are not evidence, and long campaigns overlap, so SE must vary with campaign length.""" + table = calibration_tables(_cases([6.0]), n_replicates=64)["headline_by_block_and_length"] + by_len = table.set_index("campaign_weeks") + assert by_len.loc[1, "n_independent"] == 64 # short campaigns: positions are not the constraint + assert by_len.loc[1, "coverage_se"] == pytest.approx(0.058, abs=0.001) + + +class TestIndependentDraws: + def test_short_campaigns_are_capped_by_the_replicate_count(self) -> None: + """A 1-week campaign has ~200 positions in a 4-year range; the replicates are the limit.""" + assert independent_draws(1, n_replicates=64) == 64 + + def test_long_campaigns_are_limited_by_overlapping_windows(self) -> None: + """The point of this: 64 replicates of a year-long campaign are NOT 64 independent draws.""" + assert independent_draws(52, n_replicates=64) < 64 + + def test_draws_fall_as_campaigns_lengthen(self) -> None: + draws = [independent_draws(w, n_replicates=64) for w in (1, 4, 26, 52)] + assert draws == sorted(draws, reverse=True) + assert draws[-1] < draws[0] + + def test_a_campaign_longer_than_the_start_range_still_has_one_position(self) -> None: + assert independent_draws(52 * 10, n_replicates=64) >= 1 diff --git a/tests/benchmarking/baselines/test_time_features.py b/tests/benchmarking/baselines/test_time_features.py new file mode 100644 index 00000000..08789bf8 --- /dev/null +++ b/tests/benchmarking/baselines/test_time_features.py @@ -0,0 +1,145 @@ +"""Tests for the explicit time features in ``benchmarking.baselines.time_features``. + +Checks known-input values for each feature (a plain linear day-count, a solstice-anchored +sin/cos pair, and NOAA solar-position altitude/azimuth against hand-verified reference +points), plus the shared tz-naive-index error path. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.time_features import ( + TIME_FEATURE_NAMES, + days_since_campaign_start, + season_sin_cos, + solar_altitude_azimuth, +) + +# Hill of Towie, Aberdeenshire -- a real wind_up test site, used as the solar-position fixture. +_HILL_OF_TOWIE_LATITUDE = 57.50 +_HILL_OF_TOWIE_LONGITUDE = -3.25 + + +def test_time_feature_names_vocabulary() -> None: + assert TIME_FEATURE_NAMES == ("days_since_campaign_start", "season", "solar") + + +def test_days_since_campaign_start_exact_values() -> None: + campaign_start = pd.Timestamp("2020-01-01 00:00:00", tz="UTC") + index = pd.DatetimeIndex(["2019-12-31 00:00:00", "2020-01-01 00:00:00", "2020-01-02 12:00:00"], tz="UTC") + + result = days_since_campaign_start(index, campaign_start=campaign_start) + + assert result.name == "days_since_campaign_start" + np.testing.assert_allclose(result.to_numpy(), [-1.0, 0.0, 1.5]) + assert result.index.equals(index) + + +def test_days_since_campaign_start_raises_on_tz_naive_campaign_start() -> None: + index = pd.date_range("2018-06-21", periods=3, freq="D", tz="UTC") + with pytest.raises(ValueError, match="campaign_start must be timezone-aware"): + days_since_campaign_start(index, campaign_start=pd.Timestamp("2018-06-22")) + + +def test_days_since_campaign_start_raises_on_tz_naive_index() -> None: + index = pd.DatetimeIndex(["2020-01-01 00:00:00"]) + with pytest.raises(ValueError, match="timezone-aware"): + days_since_campaign_start(index, campaign_start=pd.Timestamp("2020-01-01", tz="UTC")) + + +def test_season_sin_cos_june_solstice_noon() -> None: + index = pd.DatetimeIndex(["2018-06-21 12:00:00"], tz="UTC") + + result = season_sin_cos(index) + + assert list(result.columns) == ["season_sin", "season_cos"] + np.testing.assert_allclose(result["season_cos"].to_numpy(), [1.0], atol=0.01) + np.testing.assert_allclose(result["season_sin"].to_numpy(), [0.0], atol=0.02) + + +def test_season_sin_cos_december_solstice() -> None: + index = pd.DatetimeIndex(["2018-12-21 00:00:00"], tz="UTC") + + result = season_sin_cos(index) + + np.testing.assert_allclose(result["season_cos"].to_numpy(), [-1.0], atol=0.01) + + +def test_season_sin_cos_is_on_unit_circle() -> None: + index = pd.date_range("2018-01-01", periods=50, freq="7D", tz="UTC") + + result = season_sin_cos(index) + + magnitude_sq = result["season_sin"].to_numpy() ** 2 + result["season_cos"].to_numpy() ** 2 + np.testing.assert_allclose(magnitude_sq, np.ones_like(magnitude_sq)) + + +def test_season_sin_cos_leap_year_anchor_stays_on_june21() -> None: + # 2020 is a leap year: June 21 is day-of-year 173, and the anchor must move with it. + index = pd.DatetimeIndex(["2020-06-21 12:00:00"], tz="UTC") + + result = season_sin_cos(index) + + np.testing.assert_allclose(result["season_cos"].to_numpy(), [1.0], atol=0.01) + np.testing.assert_allclose(result["season_sin"].to_numpy(), [0.0], atol=0.02) + + +def test_season_sin_cos_non_utc_index_encodes_the_same_instants() -> None: + utc = pd.DatetimeIndex(["2018-06-21 12:00:00"], tz="UTC") + same_instant_elsewhere = utc.tz_convert("Australia/Sydney") + + np.testing.assert_allclose( + season_sin_cos(same_instant_elsewhere).to_numpy(), season_sin_cos(utc).to_numpy(), atol=1e-12 + ) + + +def test_season_sin_cos_raises_on_tz_naive_index() -> None: + index = pd.DatetimeIndex(["2018-06-21 12:00:00"]) + with pytest.raises(ValueError, match="timezone-aware"): + season_sin_cos(index) + + +def test_solar_altitude_azimuth_hill_of_towie_summer_noon() -> None: + """Sun near its highest, roughly due south, at midsummer local noon.""" + index = pd.DatetimeIndex(["2018-06-21 12:00:00"], tz="UTC") + + result = solar_altitude_azimuth(index, latitude=_HILL_OF_TOWIE_LATITUDE, longitude=_HILL_OF_TOWIE_LONGITUDE) + + assert result["solar_altitude"].to_numpy()[0] == pytest.approx(55.7, abs=1.0) + # Azimuth close to due south (180 degrees): sin small, cos close to -1. + assert result["solar_azimuth_sin"].to_numpy()[0] == pytest.approx(0.1, abs=0.15) + assert result["solar_azimuth_cos"].to_numpy()[0] == pytest.approx(-1.0, abs=0.05) + + +def test_solar_altitude_azimuth_hill_of_towie_midnight_is_below_horizon() -> None: + index = pd.DatetimeIndex(["2018-06-21 00:00:00"], tz="UTC") + + result = solar_altitude_azimuth(index, latitude=_HILL_OF_TOWIE_LATITUDE, longitude=_HILL_OF_TOWIE_LONGITUDE) + + assert result["solar_altitude"].to_numpy()[0] < 0.0 + + +def test_solar_altitude_azimuth_equator_equinox_noon_near_zenith() -> None: + index = pd.DatetimeIndex(["2019-03-21 12:00:00"], tz="UTC") + + result = solar_altitude_azimuth(index, latitude=0.0, longitude=0.0) + + assert result["solar_altitude"].to_numpy()[0] > 85.0 + + +def test_solar_azimuth_sin_cos_on_unit_circle() -> None: + index = pd.date_range("2018-01-01", periods=100, freq="97min", tz="UTC") + + result = solar_altitude_azimuth(index, latitude=_HILL_OF_TOWIE_LATITUDE, longitude=_HILL_OF_TOWIE_LONGITUDE) + + magnitude_sq = result["solar_azimuth_sin"].to_numpy() ** 2 + result["solar_azimuth_cos"].to_numpy() ** 2 + np.testing.assert_allclose(magnitude_sq, np.ones_like(magnitude_sq)) + + +def test_solar_altitude_azimuth_raises_on_tz_naive_index() -> None: + index = pd.DatetimeIndex(["2018-06-21 12:00:00"]) + with pytest.raises(ValueError, match="timezone-aware"): + solar_altitude_azimuth(index, latitude=_HILL_OF_TOWIE_LATITUDE, longitude=_HILL_OF_TOWIE_LONGITUDE) diff --git a/tests/benchmarking/baselines/test_toggle_end_to_end.py b/tests/benchmarking/baselines/test_toggle_end_to_end.py new file mode 100644 index 00000000..7fbe0043 --- /dev/null +++ b/tests/benchmarking/baselines/test_toggle_end_to_end.py @@ -0,0 +1,65 @@ +"""Slow end-to-end test: the naive ratio method on a HoT-derived synthetic toggle dataset. + +Downloads the Hill of Towie v2 SCADA (Zenodo), injects a known constant-Cp uplift under a fast +20-min-on/20-min-off toggle, and scores ``NaiveRatioMethod`` and an oracle through the harness on +real data. This is the sanity that the toggle path composes end to end for a lightweight method; +it is marked ``slow`` (network download) and skipped by ``-m "not slow"``. + +v0 is deliberately *not* exercised here: a real wind_up run per campaign is far too slow for an +e2e test. The v0 integration has its own (env-gated) end-to-end test in ``test_v0_end_to_end``. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.naive_ratio import NaiveRatioMethod +from benchmarking.harness import StudyConfig, score_study +from benchmarking.harness.example_hot_study import OracleMethod +from benchmarking.synthetic import HOT_COLUMNS, ConstantCpChange +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + + +@pytest.mark.slow +def test_naive_recovers_toggle_uplift(tmp_path) -> None: # noqa: ANN001 + scada_df, _metadata_df = load_hot_scada( + start_dt=pd.Timestamp("2016-01-01", tz="UTC"), + end_dt_excl=pd.Timestamp("2018-09-01", tz="UTC"), + wtg_numbers=[1, 3, 4, 7], + ) + study = StudyConfig( + mode="toggle", + turbine_subset=["T01", "T03", "T04", "T07"], + treatment_start_range=(pd.Timestamp("2018-02-01", tz="UTC"), pd.Timestamp("2018-02-08", tz="UTC")), + min_pre_months=24, + campaign_months=[6], + toggle_period=pd.Timedelta(minutes=40), + n_replicates=1, + seed=0, + ) + + results = score_study( + scada_df, + profile=[ConstantCpChange(delta=0.05)], + methods=[ + NaiveRatioMethod( + columns=HOT_COLUMNS, + out_dir=tmp_path / "naive_runs", + ), + OracleMethod(scada_df), + ], + study=study, + profile_name="constant_cp_toggle", + ) + + oracle_row = results.loc[results["method"] == "oracle"].iloc[0] + naive_row = results.loc[results["method"] == "naive_ratio"].iloc[0] + + # the harness toggle path is wired correctly: the oracle recovers the injected truth exactly + assert abs(oracle_row["signed_error"]) < 1e-6 + # a real constant-Cp uplift is positive; the naive method recovers it to within a few pp. + assert naive_row["truth"] > 0 + assert np.isfinite(naive_row["estimate"]) + assert abs(naive_row["signed_error"]) < 0.03 diff --git a/tests/benchmarking/baselines/test_toggle_specialist.py b/tests/benchmarking/baselines/test_toggle_specialist.py new file mode 100644 index 00000000..f668d1ef --- /dev/null +++ b/tests/benchmarking/baselines/test_toggle_specialist.py @@ -0,0 +1,1047 @@ +"""Tests for the ToggleSpecialistMethod energy-ratio baseline. + +The method is light (no wind_up pipeline), so these run the real thing on small hand-built +SCADA frames. It is a toggle-only specialist: it rejects prepost inputs and always fits on the +interleaved campaign on/off blocks. The tests cover the estimator's toggle recovery, the +prepost rejection, complete-case data handling, the timebase variable, the active-power-only +rule, the written diagnostics, the always-on campaign-only restriction, and the error paths. +""" + +from __future__ import annotations + +import ast +from dataclasses import replace +from pathlib import Path + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines import toggle_specialist +from benchmarking.baselines.toggle_specialist import ( + ToggleSpecialistMethod, + _daily_segment_coverage, + _daily_segment_ratio, + _expected_per_day, + _infer_timebase, + restrict_to_campaign, +) +from benchmarking.harness.conditions import condition_bins +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.harness.toggle import build_toggle_df, resolve_toggle +from benchmarking.synthetic import ColumnSchema, ToggleSchedule, treated_mask + +# Deliberately non-v0 column names: the method is source-agnostic, so the active-power column +# is configured and the turbine column comes from the seam. Using names that are nothing like +# v0's ``DataColumns`` proves the method never reaches for wind_up's vocabulary. +_TURBINE_COL = "asset_id" +_POWER_COL = "kw" +_AVAIL_COL = "secs_avail" +# Larger than any test timebase's full period, so the (required) availability filter keeps +# every row unless a test deliberately sets a lower value. +_FULLY_AVAILABLE_SECS = 3600.0 +# The method reads active_power + availability from the schema; the other required roles are +# unused by the ratio, so name them with placeholders. +_COLUMNS = ColumnSchema( + turbine=_TURBINE_COL, + active_power=_POWER_COL, + wind_speed="ws", + wind_speed_sd="ws_sd", + gen_rpm="rpm", + availability=_AVAIL_COL, +) + + +def _index(n: int, *, freq: str = "10min", start: str = "2020-01-01") -> pd.DatetimeIndex: + return pd.date_range(start=start, periods=n, freq=freq, tz="UTC", name="timestamp") + + +def _scada(power_by_turbine: dict[str, np.ndarray], index: pd.DatetimeIndex) -> pd.DataFrame: + """Long-format SCADA: one block of rows per turbine, sharing ``index`` (fully available).""" + frames = [ + pd.DataFrame( + {_TURBINE_COL: name, _POWER_COL: np.asarray(vals, dtype=float), _AVAIL_COL: _FULLY_AVAILABLE_SECS}, + index=index, + ) + for name, vals in power_by_turbine.items() + ] + return pd.concat(frames) + + +def _recovery_scada( + index: pd.DatetimeIndex, *, treated: np.ndarray, k: float = 0.8, uplift: float = 0.05 +) -> pd.DataFrame: + """Build SCADA where test = k*ref_total in baseline and (1+uplift)*k*ref_total when treated.""" + ref1 = np.linspace(100.0, 1000.0, len(index)) + ref2 = np.linspace(50.0, 500.0, len(index)) + ref_total = ref1 + ref2 + test = k * ref_total + test = np.where(treated, test * (1.0 + uplift), test) + return _scada({"T1": test, "R1": ref1, "R2": ref2}, index) + + +def _toggle_case( + n: int = 40, *, period_min: int = 20, uplift: float = 0.05 +) -> tuple[pd.DataFrame, ToggleSchedule, np.ndarray]: + """A toggle campaign starting at the first row (so campaign-only drops nothing) with a known uplift.""" + idx = _index(n) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=period_min), start=idx[0]) + treated = np.asarray(treated_mask(idx, schedule)) + return _recovery_scada(idx, treated=treated, uplift=uplift), schedule, treated + + +class TestInferTimebase: + def test_infers_ten_minutes(self) -> None: + assert _infer_timebase(_index(20)) == pd.Timedelta(minutes=10) + + def test_infers_thirty_minutes(self) -> None: + assert _infer_timebase(_index(20, freq="30min")) == pd.Timedelta(minutes=30) + + def test_infers_from_duplicated_long_index(self) -> None: + # long format repeats each timestamp once per turbine + idx = _index(10) + doubled = idx.append(idx) + assert _infer_timebase(doubled) == pd.Timedelta(minutes=10) + + +def test_toggle_specialist_shares_no_wind_up_code() -> None: + """The method is an independent, source-native baseline: it must not import wind_up.""" + tree = ast.parse(Path(toggle_specialist.__file__).read_text()) + modules = {alias.name for node in ast.walk(tree) if isinstance(node, ast.Import) for alias in node.names} + modules |= {node.module for node in ast.walk(tree) if isinstance(node, ast.ImportFrom) and node.module} + offenders = {m for m in modules if m == "wind_up" or m.startswith("wind_up.")} + assert not offenders, f"toggle_specialist must not depend on wind_up, found imports: {offenders}" + + +class TestRecovery: + def test_toggle_recovers_known_uplift(self) -> None: + scada, schedule, _ = _toggle_case(uplift=0.03) + out = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert isinstance(out, MethodOutput) + assert out.p50_overall == pytest.approx(0.03) + assert out.p50_by_condition is None + + +# --- per-power-bin conditional reporting ------------------------------------------------------- + +_RATED_KW = 1200.0 +# Rows a bin needs before its estimate is precise enough to test for bias rather than for luck. +_MIN_RECORDS_FOR_BIAS_TEST = 50 + + +def _varying_rho_scada( + index: pd.DatetimeIndex, + *, + treated: np.ndarray, + uplift: float = 0.0, + noise_frac: float = 0.0, + seed: int = 0, +) -> pd.DataFrame: + """SCADA whose test/reference ratio **varies with power** — the case that separates the estimators. + + ``k`` (the untreated test/ref_total ratio) falls from 0.9 at low power to 0.7 at high power, as a + real test-vs-reference ratio does (different turbines, different wakes, saturation near rated). + Any per-bin estimator that assumes one global ratio will read this structure as uplift. + """ + ref1 = np.linspace(100.0, 1000.0, len(index)) + ref2 = np.linspace(50.0, 500.0, len(index)) + ref_total = ref1 + ref2 + span = (ref_total - ref_total.min()) / (ref_total.max() - ref_total.min()) + k = 0.9 - 0.2 * span + test = k * ref_total * np.where(treated, 1.0 + uplift, 1.0) + if noise_frac: + rng = np.random.default_rng(seed) + test = test * (1.0 + rng.normal(0.0, noise_frac, len(index))) + return _scada({"T1": test, "R1": ref1, "R2": ref2}, index) + + +def _varying_rho_case( + n: int = 600, *, uplift: float = 0.0, noise_frac: float = 0.0 +) -> tuple[pd.DataFrame, ToggleSchedule]: + idx = _index(n) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=idx[0]) + treated = np.asarray(treated_mask(idx, schedule)) + return _varying_rho_scada(idx, treated=treated, uplift=uplift, noise_frac=noise_frac), schedule + + +def _per_bin(scada: pd.DataFrame, schedule: ToggleSchedule) -> pd.DataFrame: + """Run the method with power conditioning on and return its populated per-bin rows.""" + out = ToggleSpecialistMethod(columns=_COLUMNS, conditions=("power",), rated_power_kw=_RATED_KW).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_by_condition is not None + frame = out.p50_by_condition + return frame[frame["n_records"] > 0] + + +class TestPerBinIsNotBiasedByTheBaselineRatio: + """The two tests that earn the per-bin `rho_base` design. + + Both rejected estimators fail here: + + - a **global** ``rho_base`` denominator reads the power-dependence of ``k`` as uplift, tilting + the per-bin curve (positive where ``k`` is above its average, negative where below); + - binning on the **test turbine's own power** labels each bin with a treated quantity, so with a + real uplift the treated rows in a bin correspond to *lower* untreated power than the baseline + rows in it — against a varying ``k`` that mismatch becomes bias. + """ + + def test_flat_zero_truth_reads_zero_in_every_bin(self) -> None: + # the placebo: k varies strongly with power but no treatment is applied, so a correct + # estimator reports 0 everywhere. A global-rho_base estimator reports a slope instead. + scada, schedule = _varying_rho_case(uplift=0.0) + per_bin = _per_bin(scada, schedule) + assert len(per_bin) >= 3 # several populated bins, or the test proves little + assert per_bin["p50_uplift"].abs().max() < 0.005 + + def test_constant_uplift_reads_that_uplift_in_every_bin(self) -> None: + # strictly stronger than the placebo: a real uplift shifts the test turbine's power, so an + # estimator that bins on that power mis-assigns treated rows relative to baseline rows. With + # k varying, that mismatch shows up as a per-bin error rather than cancelling. + scada, schedule = _varying_rho_case(uplift=0.05) + per_bin = _per_bin(scada, schedule) + assert len(per_bin) >= 3 + assert per_bin["p50_uplift"].to_numpy() == pytest.approx(0.05, abs=0.005) + + def test_holds_with_noise(self) -> None: + # this checks bias, not variance, so it needs enough rows that per-bin sampling noise cannot + # explain a miss: at 2% noise over 6 bins x 2 segments, 2000 rows puts the per-bin standard + # error near 0.2 pp, so the 1 pp bound is a genuine bias test rather than a coin flip. + # That standard error assumes a *populated* bin. The extreme bins of this fixture hold a + # handful of rows (the lowest, one), where the method itself reports a sigma of ~14 pp -- a + # 1 pp bound there tests the random number generator, not the estimator. + scada, schedule = _varying_rho_case(n=2000, uplift=0.05, noise_frac=0.02) + per_bin = _per_bin(scada, schedule) + populated = per_bin[per_bin["n_records"] >= _MIN_RECORDS_FOR_BIAS_TEST] + assert len(populated) >= 3 + assert populated["p50_uplift"].to_numpy() == pytest.approx(0.05, abs=0.01) + + +class TestPerBinLocalisesUplift: + def test_uplift_confined_to_high_power_shows_only_there(self) -> None: + # a bin-local treatment must not smear across bins: only the treated bins move. + idx = _index(600) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=idx[0]) + treated = np.asarray(treated_mask(idx, schedule)) + ref1 = np.linspace(100.0, 1000.0, len(idx)) + ref2 = np.linspace(50.0, 500.0, len(idx)) + ref_total = ref1 + ref2 + test = 0.8 * ref_total + # +10% only where the untreated test power is high + high = test > 700.0 + test = np.where(treated & high, test * 1.10, test) + scada = _scada({"T1": test, "R1": ref1, "R2": ref2}, idx) + + per_bin = _per_bin(scada, schedule) + low_bins = per_bin[per_bin["sum_counterfactual"] > 0].iloc[:2] + assert low_bins["p50_uplift"].abs().max() < 0.005 + assert per_bin["p50_uplift"].max() == pytest.approx(0.10, abs=0.01) + + +class TestConditionsConfiguration: + def test_default_reports_no_conditions(self) -> None: + # back-compat: existing callers get exactly today's behaviour and need no rating + scada, schedule, _ = _toggle_case(uplift=0.03) + out = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_by_condition is None + + def test_power_conditioning_does_not_move_the_headline(self) -> None: + # the per-bin decomposition is additional reporting, never a change to the estimate + scada, schedule, _ = _toggle_case(uplift=0.03) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + plain = ToggleSpecialistMethod(columns=_COLUMNS).estimate(mi) + conditioned = ToggleSpecialistMethod( + columns=_COLUMNS, conditions=("power",), rated_power_kw=_RATED_KW + ).estimate(mi) + assert conditioned.p50_overall == plain.p50_overall + + def test_frame_is_labelled_with_the_power_condition(self) -> None: + scada, schedule = _varying_rho_case(uplift=0.05) + out = ToggleSpecialistMethod(columns=_COLUMNS, conditions=("power",), rated_power_kw=_RATED_KW).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_by_condition is not None + assert set(out.p50_by_condition["condition"]) == {"power"} + for col in ("condition_bin", "p50_uplift", "n_records", "sum_actual", "sum_counterfactual"): + assert col in out.p50_by_condition.columns + + def test_ws_condition_raises_citing_the_method_limit(self) -> None: + with pytest.raises(ValueError, match="does not support"): + ToggleSpecialistMethod(columns=_COLUMNS, conditions=("ws",), rated_power_kw=_RATED_KW) + + def test_unknown_condition_raises(self) -> None: + with pytest.raises(ValueError, match="unknown condition"): + ToggleSpecialistMethod(columns=_COLUMNS, conditions=("bogus",), rated_power_kw=_RATED_KW) + + def test_power_without_a_rating_raises(self) -> None: + with pytest.raises(ValueError, match="rated_power_kw"): + ToggleSpecialistMethod(columns=_COLUMNS, conditions=("power",)) + + +class TestPerBinSparseData: + def test_bins_with_no_data_are_nan_not_imputed(self) -> None: + # a sparse bin must read "no answer", not an invented one: downstream consumers decide what to + # do about it, and an imputed prior would manufacture false confidence. Rating the turbine far + # above the data's power range guarantees empty upper bins rather than hoping for them. + scada, schedule = _varying_rho_case(uplift=0.05) + out = ToggleSpecialistMethod(columns=_COLUMNS, conditions=("power",), rated_power_kw=4000.0).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_by_condition is not None + empty = out.p50_by_condition[out.p50_by_condition["n_records"] == 0] + assert not empty.empty + assert empty["p50_uplift"].isna().all() + + def test_every_bin_is_represented(self) -> None: + scada, schedule = _varying_rho_case(uplift=0.05) + out = ToggleSpecialistMethod(columns=_COLUMNS, conditions=("power",), rated_power_kw=_RATED_KW).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_by_condition is not None + assert len(out.p50_by_condition) == len(condition_bins("power", rated_power_kw=_RATED_KW)) - 1 + + +class TestPerBinDiagnostics: + def test_per_bin_csv_is_written(self, tmp_path: Path) -> None: + scada, schedule = _varying_rho_case(uplift=0.05) + ToggleSpecialistMethod( + columns=_COLUMNS, conditions=("power",), rated_power_kw=_RATED_KW, out_dir=tmp_path + ).estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL)) + assert list(tmp_path.rglob("*_by_power_bin_*.csv")) + + def test_no_per_bin_csv_when_conditioning_is_off(self, tmp_path: Path) -> None: + scada, schedule, _ = _toggle_case(uplift=0.03) + ToggleSpecialistMethod(columns=_COLUMNS, out_dir=tmp_path).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert not list(tmp_path.rglob("*_by_power_bin_*.csv")) + + def test_per_bin_plot_written_with_save_plots(self, tmp_path: Path) -> None: + scada, schedule = _varying_rho_case(uplift=0.05) + ToggleSpecialistMethod( + columns=_COLUMNS, + conditions=("power",), + rated_power_kw=_RATED_KW, + out_dir=tmp_path, + save_plots=True, + ).estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL)) + assert list(tmp_path.rglob("*per_bin_uplift.png")) + + +class TestPrepostRejected: + """The specialist only supports toggle campaigns; a prepost changeover must raise.""" + + def test_prepost_timestamp_raises(self) -> None: + idx = _index(20) + upgrade = idx[10] # a bare Timestamp is a prepost changeover, not a toggle + treated = np.asarray(idx >= upgrade) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + with pytest.raises(ValueError, match="toggle"): + ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=upgrade, turbine_col=_TURBINE_COL) + ) + + +class TestDowntimeFilter: + """Downtime filtering is required and applies to the test turbine and every reference.""" + + def test_columns_is_required(self) -> None: + with pytest.raises(TypeError): + ToggleSpecialistMethod() # type: ignore[call-arg] + + @pytest.mark.parametrize("blank", ["", " "]) + def test_blank_availability_role_raises(self, blank: str) -> None: + # A schema that leaves the availability role blank (empty or whitespace) would silently skip + # downtime filtering, so construction must reject it. + with pytest.raises(ValueError, match="availability"): + ToggleSpecialistMethod(columns=replace(_COLUMNS, availability=blank)) + + def test_missing_availability_column_raises(self) -> None: + scada, schedule, _ = _toggle_case() + scada = scada.drop(columns=[_AVAIL_COL]) + with pytest.raises(ValueError, match="availability"): + ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + + def test_down_reference_timestamps_are_excluded(self) -> None: + scada, schedule, _ = _toggle_case() + idx = scada.index.unique().sort_values() + # Corrupt a reference's power at two timestamps AND mark it unavailable there. The downtime + # filter must drop those timestamps, so the wild power never biases the ratio. + corrupted = scada.copy() + down = (corrupted[_TURBINE_COL] == "R1") & corrupted.index.isin([idx[3], idx[4]]) + corrupted.loc[down, _POWER_COL] = 1e6 + corrupted.loc[down, _AVAIL_COL] = 0.0 + out = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=corrupted, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_overall == pytest.approx(0.05) + + +class TestCompleteCase: + def test_drops_timestamp_when_a_reference_is_nan(self) -> None: + scada, schedule, _ = _toggle_case() + idx = scada.index.unique().sort_values() + clean = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + + # NaN one reference at a used timestamp: that timestamp must be excluded, but since the + # ratio is identical within each segment the estimate is unchanged. + corrupted = scada.copy() + mask = (corrupted[_TURBINE_COL] == "R1") & (corrupted.index == idx[3]) + corrupted.loc[mask, _POWER_COL] = np.nan + out = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=corrupted, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_overall == pytest.approx(clean.p50_overall) + + def test_nan_at_unused_timestamp_does_not_change_estimate(self) -> None: + scada, schedule, _ = _toggle_case() + idx = scada.index.unique().sort_values() + # make idx[2] already unused (test NaN), then add a second NaN at the same timestamp + base = scada.copy() + base.loc[(base[_TURBINE_COL] == "T1") & (base.index == idx[2]), _POWER_COL] = np.nan + before = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=base, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + after_df = base.copy() + after_df.loc[(after_df[_TURBINE_COL] == "R1") & (after_df.index == idx[2]), _POWER_COL] = np.nan + after = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=after_df, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert after.p50_overall == pytest.approx(before.p50_overall) + + +class TestTimebaseInvariance: + def test_estimate_invariant_to_timebase(self) -> None: + results = [] + for freq in ("10min", "30min"): + idx = _index(40, freq=freq) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20) * (3 if freq == "30min" else 1), start=idx[0]) + treated = np.asarray(treated_mask(idx, schedule)) + scada = _recovery_scada(idx, treated=treated, uplift=0.04) + results.append( + ToggleSpecialistMethod(columns=_COLUMNS) + .estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL)) + .p50_overall + ) + assert results[0] == pytest.approx(results[1]) + + def test_override_timebase_used_for_mwh(self, tmp_path) -> None: # noqa: ANN001 + scada, schedule, _ = _toggle_case() + method = ToggleSpecialistMethod( + columns=_COLUMNS, + out_dir=tmp_path, + timebase=pd.Timedelta(minutes=30), + ) + method.estimate(MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL)) + stats = _read_only_csv(tmp_path, "data_stats") + all_row = stats[stats["segment"] == "all"].iloc[0] + # MWh = sum(power_kw) * timebase_hours / 1000; check it used 0.5h not 1/6h + expected = all_row["used_test_mean_power_kw"] * all_row["n_used_timestamps"] * 0.5 / 1000.0 + assert all_row["used_test_mwh"] == pytest.approx(expected) + + +class TestActivePowerOnly: + def test_extra_columns_ignored(self) -> None: + scada, schedule, _ = _toggle_case() + plain = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + + with_extra = scada.copy() + rng = np.random.default_rng(0) + with_extra["ws"] = rng.normal(size=len(with_extra)) + with_extra["rpm"] = rng.normal(size=len(with_extra)) + out = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=with_extra, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.p50_overall == pytest.approx(plain.p50_overall) + + +class TestErrors: + def test_no_reference_raises(self) -> None: + idx = _index(40) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=idx[0]) + scada = _scada({"T1": np.ones(len(idx))}, idx) + with pytest.raises(ValueError, match="reference"): + ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + + def test_no_used_baseline_returns_nan(self) -> None: + scada, schedule, treated = _toggle_case() + idx = scada.index.unique().sort_values() + # NaN the test turbine across the whole off (baseline) class -> no used baseline timestamps + off_ts = idx[~treated] + is_baseline_test = (scada[_TURBINE_COL] == "T1") & scada.index.isin(off_ts) + scada.loc[is_baseline_test, _POWER_COL] = np.nan + out = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert np.isnan(out.p50_overall) + + +class TestPlotSeries: + """The per-segment daily series that back the ratio plot and the coverage plot.""" + + def test_ratio_series_split_by_segment(self) -> None: + # one day off (ratio 0.8) then one day on (ratio 0.8*1.05); daily sum-based ratio. + idx = _index(288, freq="10min") # two full days + treated = np.asarray(idx >= idx[144]) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + wide = scada.pivot_table(index=scada.index, columns=_TURBINE_COL, values=_POWER_COL) + test_pw = wide["T1"].to_numpy() + ref_total = wide[["R1", "R2"]].sum(axis=1).to_numpy() + used = np.ones(len(wide), dtype=bool) + + base = _daily_segment_ratio(wide.index, test_pw, ref_total, used & ~treated) + up = _daily_segment_ratio(wide.index, test_pw, ref_total, used & treated) + # each segment only has data on its own day; the other day is NaN. + assert base.dropna().to_numpy() == pytest.approx(0.8) + assert up.dropna().to_numpy() == pytest.approx(0.8 * 1.05) + assert base.index.equals(up.index) + + def test_toggle_coverage_capped_near_duty_cycle(self) -> None: + # 20-on/20-off toggle on 10-min data -> each segment can occupy at most ~50% of a day. + idx = _index(288, freq="10min") + schedule = ToggleSchedule(period=pd.Timedelta(minutes=40), start=idx[0]) + treated = np.asarray(treated_mask(idx, schedule)) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + wide = scada.pivot_table(index=scada.index, columns=_TURBINE_COL, values=_POWER_COL) + used = np.ones(len(wide), dtype=bool) + + expected = _expected_per_day(wide.index, _infer_timebase(wide.index)) + base = _daily_segment_coverage(wide.index, used, ~treated, expected) + up = _daily_segment_coverage(wide.index, used, treated, expected) + assert base.to_numpy() == pytest.approx(0.5) + assert up.to_numpy() == pytest.approx(0.5) + + +def _read_only_csv(folder: Path, kind: str) -> pd.DataFrame: + run_dirs = [p for p in Path(folder).iterdir() if p.is_dir()] + assert len(run_dirs) == 1, f"expected exactly one run dir, found {run_dirs}" + matches = list(run_dirs[0].glob(f"*_{kind}_*.csv")) + assert len(matches) == 1, f"expected one {kind} csv, found {matches}" + return pd.read_csv(matches[0]) + + +class TestDiagnostics: + def _run(self, tmp_path, *, save_plots: bool = False): # noqa: ANN001, ANN202 + scada, schedule, _ = _toggle_case(uplift=0.06) + method = ToggleSpecialistMethod(columns=_COLUMNS, out_dir=tmp_path, save_plots=save_plots) + out = method.estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + return out, schedule + + def test_writes_both_csvs(self, tmp_path) -> None: # noqa: ANN001 + self._run(tmp_path) + stats = _read_only_csv(tmp_path, "data_stats") + results = _read_only_csv(tmp_path, "results") + assert sorted(stats["segment"]) == ["all", "baseline", "upgraded"] + assert len(results) == 1 + + def test_uplift_rederivable_from_stats(self, tmp_path) -> None: # noqa: ANN001 + out, _ = self._run(tmp_path) + stats = _read_only_csv(tmp_path, "data_stats").set_index("segment") + rho_base = stats.loc["baseline", "used_test_mwh"] / stats.loc["baseline", "used_ref_total_mwh"] + rho_up = stats.loc["upgraded", "used_test_mwh"] / stats.loc["upgraded", "used_ref_total_mwh"] + assert (rho_up / rho_base - 1.0) == pytest.approx(out.p50_overall) + + def test_results_csv_matches_estimate(self, tmp_path) -> None: # noqa: ANN001 + out, _ = self._run(tmp_path) + results = _read_only_csv(tmp_path, "results").iloc[0] + assert results["uplift_frc"] == pytest.approx(out.p50_overall) + assert results["mode"] == "toggle" + assert results["n_refs"] == 2 + + def test_stats_coverage_full_for_complete_data(self, tmp_path) -> None: # noqa: ANN001 + self._run(tmp_path) + stats = _read_only_csv(tmp_path, "data_stats").set_index("segment") + assert stats.loc["all", "rows_data_coverage"] == pytest.approx(1.0) + assert stats.loc["all", "used_data_coverage"] == pytest.approx(1.0) + assert stats.loc["all", "n_used_timestamps"] == 40 + + def test_save_plots_writes_pngs(self, tmp_path) -> None: # noqa: ANN001 + self._run(tmp_path, save_plots=True) + run_dir = next(p for p in Path(tmp_path).iterdir() if p.is_dir()) + names = {p.name for p in (run_dir / "plots").rglob("*.png")} + # the method's own plots, plus the shared cross-method diagnostics it now emits + assert {"T1_scatter.png", "T1_ratio_timeseries.png", "T1_coverage_timeseries.png"} <= names + # a run-config YAML is written alongside the plots + assert any(run_dir.glob("config_*.yaml")) + + def test_no_plots_by_default(self, tmp_path) -> None: # noqa: ANN001 + self._run(tmp_path) + run_dir = next(p for p in Path(tmp_path).iterdir() if p.is_dir()) + assert not (run_dir / "plots").exists() + + +class TestCampaignOnly: + """The campaign-only restriction is mandatory: a pre-campaign window never enters the baseline.""" + + def test_precampaign_dropped_from_baseline(self, tmp_path) -> None: # noqa: ANN001 + idx = _index(300) + start = idx[100] # 100 pre-campaign rows, then 200 rows of interleaved on/off + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=start) + treated = np.asarray(treated_mask(idx, schedule)) + scada = _recovery_scada(idx, treated=treated, uplift=0.05) + ToggleSpecialistMethod(columns=_COLUMNS, out_dir=tmp_path).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + baseline = _read_only_csv(tmp_path, "data_stats").set_index("segment").loc["baseline"] + # the baseline class starts at/after the campaign start; the pre-campaign rows are excluded. + assert pd.Timestamp(baseline["first_timestamp"]) >= start + + +# --- uncertainty (non-optional, computed after the uplift) -------------------------------------- + + +def _noisy_toggle_case(n: int = 2016, *, uplift: float = 0.03, noise_frac: float = 0.05) -> tuple: + """A campaign long enough to bootstrap, with per-record noise so sigma has something to find.""" + idx = _index(n) + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=idx[0]) + treated = np.asarray(treated_mask(idx, schedule)) + scada = _varying_rho_scada(idx, treated=treated, uplift=uplift, noise_frac=noise_frac) + return scada, schedule + + +def _estimate(scada: pd.DataFrame, schedule: ToggleSchedule, **kwargs: object) -> MethodOutput: + return ToggleSpecialistMethod(columns=_COLUMNS, **kwargs).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + + +class TestUncertaintyIsAlwaysReported: + def test_overall_sigma_is_reported_without_being_asked_for(self) -> None: + """Uncertainty is not an opt-in: an uplift with no sigma is not a usable answer.""" + out = _estimate(*_noisy_toggle_case()) + assert out.sigma_overall is not None + assert np.isfinite(out.sigma_overall) + assert out.sigma_overall > 0 + + def test_every_reported_bin_carries_a_sigma(self) -> None: + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.p50_by_condition is not None + assert "sigma_uplift" in out.p50_by_condition.columns + populated = out.p50_by_condition[out.p50_by_condition["n_records"] > 0] + assert len(populated) > 0 + assert populated["sigma_uplift"].notna().all() + + def test_diagnostics_cover_the_headline_and_every_bin(self) -> None: + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + diag = out.uncertainty_diagnostics + assert diag is not None + assert (diag["condition"] == "overall").sum() == 1 + assert set(diag.columns) >= { + "condition", + "condition_bin", + "n_upgraded_records", + "n_baseline_records", + "n_blocks", + "sigma_robust", + "frac_resamples_finite", + } + overall = diag[diag["condition"] == "overall"].iloc[0] + assert overall["n_upgraded_records"] > 0 + assert overall["n_baseline_records"] > 0 + + +class TestUncertaintyDoesNotChangeUplift: + def test_block_length_moves_the_bootstrap_and_nothing_else(self) -> None: + """The session's hard constraint, at the method's own seam. + + Asserted on the *bootstrap* component, not the reported sigma: the reported value is + ``max(bootstrap, fallback)``, and the fallback does not depend on block length — so on data + where it wins at both lengths it would mask the difference this test exists to check. + """ + scada, schedule = _noisy_toggle_case() + a = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW, block_hours=6.0) + b = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW, block_hours=96.0) + assert a.p50_overall == b.p50_overall + pd.testing.assert_series_equal( + a.p50_by_condition["p50_uplift"], + b.p50_by_condition["p50_uplift"], # type: ignore[index] + ) + boots = [ + out.uncertainty_diagnostics.set_index("condition_bin").loc["overall", "sigma_bootstrap"] # type: ignore[union-attr] + for out in (a, b) + ] + assert boots[0] != boots[1] + + def test_uplift_matches_a_run_with_the_bootstrap_reduced_to_nothing(self) -> None: + scada, schedule = _noisy_toggle_case() + full = _estimate(scada, schedule, n_resamples=1000) + minimal = _estimate(scada, schedule, n_resamples=2) + assert full.p50_overall == minimal.p50_overall + + +class TestUncertaintyRunsOnlyWhenThereIsAnUpliftToQualify: + def test_a_nan_uplift_reports_a_nan_sigma(self) -> None: + """No baseline rows means no uplift; there is nothing for an uncertainty to describe.""" + idx = _index(40) + # every row treated -> no off rows -> rho_base is NaN + scada = _recovery_scada(idx, treated=np.ones(len(idx), dtype=bool)) + out = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput( + scada_df=scada, + test_wtg="T1", + upgrade_timing=pd.DataFrame({"toggle_on": True, "toggle_off": False}, index=idx), + turbine_col=_TURBINE_COL, + ) + ) + assert np.isnan(out.p50_overall) + assert out.sigma_overall is not None + assert np.isnan(out.sigma_overall) + + def test_the_bootstrap_is_not_run_when_the_uplift_is_nan(self) -> None: + """Diagnostics still report the counts: they are what explains why there was no answer.""" + idx = _index(40) + scada = _recovery_scada(idx, treated=np.ones(len(idx), dtype=bool)) + out = ToggleSpecialistMethod(columns=_COLUMNS, conditions=("power",), rated_power_kw=_RATED_KW).estimate( + MethodInput( + scada_df=scada, + test_wtg="T1", + upgrade_timing=pd.DataFrame({"toggle_on": True, "toggle_off": False}, index=idx), + turbine_col=_TURBINE_COL, + ) + ) + diag = out.uncertainty_diagnostics + assert diag is not None + assert (diag["n_blocks"] == 0).all() + assert diag["frac_resamples_finite"].isna().all() + assert (diag["n_baseline_records"] == 0).all() + + +class TestUncertaintyReproducibility: + def test_the_same_seed_gives_the_same_sigma(self) -> None: + scada, schedule = _noisy_toggle_case() + a = _estimate(scada, schedule, bootstrap_seed=3) + b = _estimate(scada, schedule, bootstrap_seed=3) + assert a.sigma_overall == b.sigma_overall + + def test_a_longer_campaign_reports_a_smaller_sigma(self) -> None: + short = _estimate(*_noisy_toggle_case(n=1008)) + long = _estimate(*_noisy_toggle_case(n=8064)) + assert long.sigma_overall < short.sigma_overall # type: ignore[operator] + + +class TestUncertaintyInTheWrittenOutputs: + def test_results_csv_records_the_sigma_and_the_bootstrap_config(self, tmp_path: Path) -> None: + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, out_dir=tmp_path, block_hours=24.0) + results = pd.concat([pd.read_csv(p) for p in tmp_path.glob("*/*_results_*.csv")]) + assert results["uplift_sigma_frc"].iloc[0] == pytest.approx(out.sigma_overall) + assert results["block_hours"].iloc[0] == 24.0 + assert results["n_resamples"].iloc[0] == 1000 + + def test_an_uncertainty_csv_is_written(self, tmp_path: Path) -> None: + scada, schedule = _noisy_toggle_case() + _estimate(scada, schedule, out_dir=tmp_path, conditions=("power",), rated_power_kw=_RATED_KW) + written = list(tmp_path.glob("*/*_uncertainty_*.csv")) + assert len(written) == 1 + frame = pd.read_csv(written[0]) + assert "n_upgraded_records" in frame.columns + assert len(frame) == 7 # overall + six power bins + + +# --- labeled_rows: the row selection the estimate actually used --------------------------------- + + +class TestLabeledRows: + """``labeled_rows`` exposes the method's own used/segment/bin labels, per test-turbine record. + + A consumer that wants a per-bin quantity the method does not report (a mean pitch, say) must be + able to compute it over *exactly* the rows and bins the uplift used. These tests pin that the + labels reproduce the method's internal masks rather than a plausible re-derivation of them. + """ + + def test_labeled_rows_is_populated_for_a_toggle_campaign(self) -> None: + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.labeled_rows is not None + assert set(out.labeled_rows.columns) >= {"used", "segment", "power_bin"} + + def test_carries_the_original_test_turbine_scada_columns(self) -> None: + """The consumer aggregates the source columns, so they must survive unmodified.""" + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.labeled_rows is not None + test_rows = scada[scada[_TURBINE_COL] == "T1"] + for col in test_rows.columns: + assert col in out.labeled_rows.columns + pd.testing.assert_series_equal(out.labeled_rows[_POWER_COL], test_rows[_POWER_COL]) + + def test_one_row_per_test_turbine_record(self) -> None: + """Not the long frame, and not the references: exactly the test turbine's own rows.""" + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.labeled_rows is not None + assert len(out.labeled_rows) == (scada[_TURBINE_COL] == "T1").sum() + assert out.labeled_rows.index.is_unique + + def test_used_reproduces_the_methods_own_used_mask(self) -> None: + scada, schedule = _noisy_toggle_case() + method = ToggleSpecialistMethod(columns=_COLUMNS, conditions=("power",), rated_power_kw=_RATED_KW) + mi = MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + out = method.estimate(mi) + assert out.labeled_rows is not None + + mi_restricted = restrict_to_campaign(mi) + wide = toggle_specialist._wide_column( # noqa: SLF001 + mi_restricted.scada_df, turbine_col=_TURBINE_COL, value_col=_POWER_COL + ) + expected = method._used_mask( # noqa: SLF001 + mi_restricted, + wide=wide, + test="T1", + refs=[c for c in wide.columns if c != "T1"], + timebase=pd.Timedelta(minutes=10), + ) + assert out.labeled_rows["used"].to_numpy().tolist() == expected.to_numpy().tolist() + + def test_a_downtime_row_is_labelled_unused(self) -> None: + """The label tracks the filter: knock one record's availability out and it drops out of `used`.""" + scada, schedule = _noisy_toggle_case() + victim = scada.index[scada[_TURBINE_COL] == "T1"][100] + scada.loc[(scada.index == victim) & (scada[_TURBINE_COL] == "T1"), _AVAIL_COL] = 0.0 + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.labeled_rows is not None + assert not out.labeled_rows.loc[victim, "used"] + assert out.labeled_rows["used"].sum() > 0 + + def test_segment_partitions_rows_into_baseline_upgraded_excluded(self) -> None: + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.labeled_rows is not None + assert set(out.labeled_rows["segment"]) <= {"baseline", "upgraded", "excluded"} + + def test_segment_reproduces_the_resolved_toggle_rows(self) -> None: + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.labeled_rows is not None + rows = resolve_toggle(schedule, pd.DatetimeIndex(out.labeled_rows.index)) + assert (out.labeled_rows["segment"] == "upgraded").to_numpy().tolist() == list(rows.upgraded) + assert (out.labeled_rows["segment"] == "baseline").to_numpy().tolist() == list(rows.campaign_baseline) + + def test_power_bin_uses_the_same_edges_as_the_uplift(self) -> None: + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.labeled_rows is not None + assert out.p50_by_condition is not None + labelled = set(out.labeled_rows["power_bin"].dropna().astype(str)) + reported = set(out.p50_by_condition["condition_bin"].astype(str)) + assert labelled <= reported + + def test_upgraded_rows_per_bin_match_the_reported_counts(self) -> None: + """The point of the frame: aggregate it and you land on the method's own population. + + ``n_records`` counts the *upgraded* rows a bin's ratio consumed, so that is what the labels + must reproduce -- for every bin the method returned a finite uplift for. + """ + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW) + assert out.labeled_rows is not None + assert out.p50_by_condition is not None + + rows = out.labeled_rows + counted = ( + rows[rows["used"] & (rows["segment"] == "upgraded")] + .groupby(rows["power_bin"].astype(str), observed=True) + .size() + ) + reported = out.p50_by_condition[np.isfinite(out.p50_by_condition["p50_uplift"])] + assert not reported.empty + for _, row in reported.iterrows(): + assert counted.get(str(row["condition_bin"]), 0) == row["n_records"] + + def test_rows_outside_the_bin_edges_have_no_bin(self) -> None: + """A row off the end of the bins is NaN, not silently folded into the edge bin. + + The bin edges scale with the declared rating, so rating the turbine well below the power the + fixture actually reaches pushes its top rows past the outermost edge. + """ + scada, schedule = _noisy_toggle_case() + out = _estimate(scada, schedule, conditions=("power",), rated_power_kw=_RATED_KW / 4) + assert out.labeled_rows is not None + assert out.labeled_rows["power_bin"].isna().any() + assert out.labeled_rows["power_bin"].notna().any() + + +# --- reversal symmetry ------------------------------------------------------------------------- + + +def _swap_states(toggle_df: pd.DataFrame) -> pd.DataFrame: + """Relabel which state is 'on': the same campaign, described from the other side.""" + return toggle_df.rename(columns={"toggle_on": "toggle_off", "toggle_off": "toggle_on"}) + + +class TestReversalSymmetry: + """Which of two toggle states is *called* the baseline is a naming choice, not a measurement. + + The bin label must therefore not depend on it. It used to: the label was the baseline state's + predicted power, so renaming the states shifted every label by the uplift and migrated rows + across bin edges -- moving per-bin estimates by an appreciable fraction of their own sigma for + no physical reason. + """ + + def _pair(self) -> tuple[MethodOutput, MethodOutput]: + scada, schedule = _varying_rho_case(n=600, uplift=0.05) + index = pd.DatetimeIndex(pd.unique(scada.index)).sort_values() + toggle_df = build_toggle_df(index, schedule) + forward = _estimate(scada, toggle_df, conditions=("power",), rated_power_kw=_RATED_KW) + reversed_ = _estimate(scada, _swap_states(toggle_df), conditions=("power",), rated_power_kw=_RATED_KW) + return forward, reversed_ + + def test_every_row_keeps_its_power_bin_when_the_states_are_swapped(self) -> None: + forward, reversed_ = self._pair() + assert forward.labeled_rows is not None + assert reversed_.labeled_rows is not None + fwd_bins = forward.labeled_rows["power_bin"].astype(str) + rev_bins = reversed_.labeled_rows["power_bin"].astype(str) + assert (fwd_bins == rev_bins).all(), f"{(fwd_bins != rev_bins).sum()} rows changed bin under reversal" + + def test_the_segments_simply_trade_places(self) -> None: + """Corroborates the bins are fixed: each bin's baseline rows become its upgraded rows.""" + forward, reversed_ = self._pair() + assert forward.labeled_rows is not None + assert reversed_.labeled_rows is not None + fwd, rev = forward.labeled_rows, reversed_.labeled_rows + assert (fwd["used"] == rev["used"]).all() + assert (fwd["segment"] == "upgraded").sum() == (rev["segment"] == "baseline").sum() + assert ((fwd["segment"] == "upgraded") == (rev["segment"] == "baseline")).all() + + def test_the_same_bins_are_reported_either_way(self) -> None: + forward, reversed_ = self._pair() + assert forward.p50_by_condition is not None + assert reversed_.p50_by_condition is not None + fwd = forward.p50_by_condition.set_index(forward.p50_by_condition["condition_bin"].astype(str)) + rev = reversed_.p50_by_condition.set_index(reversed_.p50_by_condition["condition_bin"].astype(str)) + assert set(fwd.index) == set(rev.index) + + +# --- caller-supplied row exclusion (ColumnSchema.exclude_row) ----------------------------------- + +_EXCLUDE_COL = "special_mode" +_EXCLUDE_COLUMNS = replace(_COLUMNS, exclude_row=_EXCLUDE_COL) + + +def _flag(scada: pd.DataFrame, *, turbine: str, every: int, value: object = True) -> pd.DataFrame: + """Return a copy of ``scada`` carrying an all-False exclude column with every Nth ``turbine`` row set.""" + out = scada.copy() + # object dtype so a non-bool ``value`` (the NaN case) can be written without an upcast warning + out[_EXCLUDE_COL] = pd.Series(data=False, index=out.index, dtype=object if value is not True else bool) + is_turbine = (out[_TURBINE_COL] == turbine).to_numpy() + position = np.cumsum(is_turbine) - 1 + out.loc[is_turbine & (position % every == 0), _EXCLUDE_COL] = value + return out + + +class TestExcludeRow: + """``columns.exclude_row`` drops caller-flagged **test** turbine rows from the estimate.""" + + def test_flagged_test_rows_are_not_used(self) -> None: + scada, schedule = _noisy_toggle_case() + flagged = _flag(scada, turbine="T1", every=5) + out = ToggleSpecialistMethod(columns=_EXCLUDE_COLUMNS).estimate( + MethodInput(scada_df=flagged, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.labeled_rows is not None + excluded = out.labeled_rows[_EXCLUDE_COL].to_numpy(dtype=bool) + assert excluded.any() + assert not out.labeled_rows.loc[excluded, "used"].to_numpy().any() + assert out.labeled_rows.loc[~excluded, "used"].to_numpy().any() + + def test_excluding_corrupted_rows_restores_the_known_uplift(self) -> None: + """The point of the feature: flagged special-mode rows stop biasing the ratio.""" + scada, schedule, _ = _toggle_case(n=400, uplift=0.05) + corrupted = _flag(scada, turbine="T1", every=4) + spoiled = corrupted[_EXCLUDE_COL].to_numpy(dtype=bool) & (corrupted[_TURBINE_COL] == "T1").to_numpy() + corrupted.loc[spoiled, _POWER_COL] *= 0.5 + + naive = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=corrupted, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + filtered = ToggleSpecialistMethod(columns=_EXCLUDE_COLUMNS).estimate( + MethodInput(scada_df=corrupted, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert filtered.p50_overall == pytest.approx(0.05, abs=1e-9) + assert abs(naive.p50_overall - 0.05) > abs(filtered.p50_overall - 0.05) + + def test_flagged_reference_rows_are_kept(self) -> None: + """References are never excluded here: their special modes still carry ratio information.""" + scada, schedule = _noisy_toggle_case() + flagged = _flag(scada, turbine="R1", every=3) + baseline = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=flagged, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + with_role = ToggleSpecialistMethod(columns=_EXCLUDE_COLUMNS).estimate( + MethodInput(scada_df=flagged, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert with_role.p50_overall == pytest.approx(baseline.p50_overall) + assert with_role.labeled_rows is not None + assert with_role.labeled_rows["used"].to_numpy().any() + + def test_nan_in_the_exclude_column_is_rejected(self) -> None: + """NaN is not "excluded": the schema contract says the column is never NaN, so say so loudly.""" + scada, schedule = _noisy_toggle_case() + flagged = _flag(scada, turbine="T1", every=7, value=np.nan) + with pytest.raises(ValueError, match=_EXCLUDE_COL): + ToggleSpecialistMethod(columns=_EXCLUDE_COLUMNS).estimate( + MethodInput(scada_df=flagged, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + + def test_unset_role_excludes_nothing(self) -> None: + scada, schedule = _noisy_toggle_case() + flagged = _flag(scada, turbine="T1", every=5) + out = ToggleSpecialistMethod(columns=_COLUMNS).estimate( + MethodInput(scada_df=flagged, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.labeled_rows is not None + excluded = out.labeled_rows[_EXCLUDE_COL].to_numpy(dtype=bool) + assert out.labeled_rows.loc[excluded, "used"].to_numpy().any() + + def test_absent_column_excludes_nothing(self) -> None: + """The role names a column the frame does not carry: skip, do not raise.""" + scada, schedule = _noisy_toggle_case() + out = ToggleSpecialistMethod(columns=_EXCLUDE_COLUMNS).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + assert out.labeled_rows is not None + assert out.labeled_rows["used"].to_numpy().any() + + def test_exclusions_reach_the_shared_diagnostics(self, tmp_path: Path) -> None: + """The exclusion mask is handed to the shared diagnostics rather than plotted bespokely. + + The 2x3 operating-curve view of the same mask needs a wind-speed column this fixture does + not carry, so it is exercised in the diagnostics tests; here the timeline is the evidence + that ``excluded_ts`` is threaded through. + """ + scada, schedule = _noisy_toggle_case() + flagged = _flag(scada, turbine="T1", every=5) + ToggleSpecialistMethod(columns=_EXCLUDE_COLUMNS, out_dir=tmp_path, save_plots=True).estimate( + MethodInput(scada_df=flagged, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + run_dir = next(p for p in Path(tmp_path).iterdir() if p.is_dir()) + names = {p.name for p in (run_dir / "plots").rglob("*.png")} + assert "excluded_row_fraction.png" in names + assert "T1_excluded_rows.png" not in names, "the bespoke plot was replaced by the shared views" + + def test_no_exclusion_plots_when_nothing_is_excluded(self, tmp_path: Path) -> None: + scada, schedule = _noisy_toggle_case() + ToggleSpecialistMethod(columns=_COLUMNS, out_dir=tmp_path, save_plots=True).estimate( + MethodInput(scada_df=scada, test_wtg="T1", upgrade_timing=schedule, turbine_col=_TURBINE_COL) + ) + run_dir = next(p for p in Path(tmp_path).iterdir() if p.is_dir()) + names = {p.name for p in (run_dir / "plots").rglob("*.png")} + assert "ops_curves_excluded.png" not in names + assert "excluded_row_fraction.png" not in names diff --git a/tests/benchmarking/baselines/test_v0_binned.py b/tests/benchmarking/baselines/test_v0_binned.py new file mode 100644 index 00000000..f2b76a6b --- /dev/null +++ b/tests/benchmarking/baselines/test_v0_binned.py @@ -0,0 +1,231 @@ +"""Offline tests for the V0BinnedMethod adapter. + +The full wind_up pipeline (``AssessmentInputs.from_cfg`` / ``run_wind_up_analysis`` / +``combine_results``) is stubbed; these tests cover the seam: the per-campaign config the +adapter builds via ``WindUpConfig.from_yaml``, the P50 extraction, and the prepost-only guard. +A real end-to-end run lives in the ``slow`` integration test. +""" + +from __future__ import annotations + +import pandas as pd +import pytest + +from benchmarking.baselines import v0_binned +from benchmarking.baselines.hot_context import HotV0Context +from benchmarking.baselines.v0_binned import V0BinnedMethod, _extract_p50, _subset_turbines +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.harness.toggle import build_toggle_df +from benchmarking.synthetic import HOT_COLUMNS, ToggleSchedule, treated_mask +from benchmarking.synthetic.sources.hill_of_towie import long_to_wind_up_format + +# The harness hands v0 source-native SCADA, so the fixtures carry the source-native tag names and +# the full field set the v0 on-ramp (``long_to_wind_up_format``) needs (pitch blades, yaw, the +# time-ready signal) to derive ``PitchAngleMean`` / ``ShutdownDuration``. +_SOURCE_FIELDS = ( + HOT_COLUMNS.active_power, + HOT_COLUMNS.wind_speed, + HOT_COLUMNS.wind_speed_sd, + HOT_COLUMNS.gen_rpm, + "wtc_ActPower_stddev", + "wtc_NacelPos_mean", + "wtc_PitcPosA_mean", + "wtc_PitcPosB_mean", + "wtc_PitcPosC_mean", +) + + +def _long_scada(turbines: list[str], index: pd.DatetimeIndex) -> pd.DataFrame: + """A source-native long SCADA frame complete enough for the v0 on-ramp to convert.""" + frames = [ + pd.DataFrame( + {HOT_COLUMNS.turbine: t, **dict.fromkeys(_SOURCE_FIELDS, 1.0), "wtc_ScReToOp_timeon": 600.0}, + index=index, + ) + for t in turbines + ] + return pd.concat(frames) + + +def _make_scada(turbines: list[str], *, start: pd.Timestamp, end: pd.Timestamp) -> pd.DataFrame: + """Minimal-span scada (two timestamps) whose meaningful content is the turbines and span.""" + index = pd.DatetimeIndex([start, end], name="TimeStamp_StartFormat") + return _long_scada(turbines, index) + + +def _context() -> HotV0Context: + return HotV0Context(metadata_df=pd.DataFrame({"Name": ["T01"]}), reanalysis_datasets=[]) + + +UPGRADE = pd.Timestamp("2018-01-01", tz="UTC") + + +def _method_input(turbines: list[str], test_wtg: str) -> MethodInput: + scada = _make_scada(turbines, start=UPGRADE - pd.DateOffset(years=1), end=UPGRADE + pd.DateOffset(months=6)) + return MethodInput(scada_df=scada, test_wtg=test_wtg, upgrade_timing=UPGRADE) + + +class TestSubsetTurbines: + def test_returns_sorted_unique_names(self) -> None: + scada = _make_scada(["T04", "T01", "T03"], start=UPGRADE, end=UPGRADE + pd.DateOffset(months=1)) + assert _subset_turbines(scada, HOT_COLUMNS.turbine) == ["T01", "T03", "T04"] + + +class TestBuildConfig: + def test_sets_test_and_ref_wtgs(self, tmp_path) -> None: # noqa: ANN001 + method = V0BinnedMethod(_context(), scratch_dir=tmp_path) + cfg = method._build_config(_method_input(["T01", "T02", "T03", "T04"], "T02")) # noqa: SLF001 + assert [w.name for w in cfg.test_wtgs] == ["T02"] + assert sorted(w.name for w in cfg.ref_wtgs) == ["T01", "T03", "T04"] + + def test_sets_prepost_dates_and_knobs(self, tmp_path) -> None: # noqa: ANN001 + method = V0BinnedMethod(_context(), scratch_dir=tmp_path) + mi = _method_input(["T01", "T02", "T03", "T04"], "T01") + cfg = method._build_config(mi) # noqa: SLF001 + assert cfg.prepost is not None + assert pd.Timestamp(cfg.prepost.post_first_dt_utc_start) == UPGRADE + assert pd.Timestamp(cfg.prepost.post_last_dt_utc_start) == pd.Timestamp(mi.scada_df.index.max()) + assert pd.Timestamp(cfg.upgrade_first_dt_utc_start) == UPGRADE + assert cfg.use_lt_distribution is False + assert cfg.optimize_northing_corrections is False + assert len(cfg.northing_corrections_utc) > 0 + + def test_filters_asset_to_subset(self, tmp_path) -> None: # noqa: ANN001 + method = V0BinnedMethod(_context(), scratch_dir=tmp_path) + cfg = method._build_config(_method_input(["T01", "T02", "T03", "T04"], "T01")) # noqa: SLF001 + assert sorted(w.name for w in cfg.asset.wtgs) == ["T01", "T02", "T03", "T04"] + + def test_raises_when_no_reference_turbines(self, tmp_path) -> None: # noqa: ANN001 + method = V0BinnedMethod(_context(), scratch_dir=tmp_path) + with pytest.raises(ValueError, match="no reference turbines"): + method._build_config(_method_input(["T01"], "T01")) # noqa: SLF001 + + +def _dense_scada(turbines: list[str], *, start: pd.Timestamp, end: pd.Timestamp, freq: str = "1D") -> pd.DataFrame: + """Source-native long scada on a regular grid (enough rows for a toggle split).""" + index = pd.date_range(start, end, freq=freq, tz="UTC", name="TimeStamp_StartFormat") + return _long_scada(turbines, index) + + +class TestBuildToggleDf: + def test_before_start_both_false_after_exactly_one_true(self) -> None: + idx = pd.date_range(UPGRADE - pd.Timedelta(minutes=30), periods=9, freq="10min", tz="UTC") + scada = _dense_scada(["a", "b"], start=idx[0], end=idx[-1], freq="10min") + schedule = ToggleSchedule(period=pd.Timedelta(minutes=20), start=UPGRADE) + df = build_toggle_df(scada.index, schedule) + + before = df.index < UPGRADE + assert not df.loc[before, "toggle_on"].any() + assert not df.loc[before, "toggle_off"].any() + after = df.index >= UPGRADE + assert (df.loc[after, "toggle_on"] ^ df.loc[after, "toggle_off"]).all() + assert (df["toggle_on"].to_numpy() == treated_mask(df.index, schedule)).all() + + +class TestBuildConfigToggle: + def test_sets_toggle_block_not_prepost(self, tmp_path) -> None: # noqa: ANN001 + method = V0BinnedMethod(_context(), scratch_dir=tmp_path) + scada = _dense_scada( + ["T01", "T02", "T03", "T04"], start=UPGRADE - pd.DateOffset(years=1), end=UPGRADE + pd.DateOffset(months=6) + ) + schedule = ToggleSchedule(period=pd.Timedelta(days=2), start=UPGRADE) + cfg = method._build_config(MethodInput(scada_df=scada, test_wtg="T01", upgrade_timing=schedule)) # noqa: SLF001 + assert cfg.toggle is not None + assert cfg.prepost is None + assert cfg.toggle.detrend_data_selection == "use_toggle_off_data" + assert cfg.toggle.toggle_change_settling_filter_seconds == 0 + assert pd.Timestamp(cfg.upgrade_first_dt_utc_start) == UPGRADE + + +class TestEstimateToggle: + def test_wires_toggle_df_and_returns_p50(self, tmp_path, monkeypatch) -> None: # noqa: ANN001 + captured = _stub_pipeline(monkeypatch) + method = V0BinnedMethod(_context(), scratch_dir=tmp_path) + scada = _dense_scada( + ["T01", "T02", "T03"], start=UPGRADE - pd.DateOffset(months=6), end=UPGRADE + pd.DateOffset(months=6) + ) + schedule = ToggleSchedule(period=pd.Timedelta(days=2), start=UPGRADE) + out = method.estimate(MethodInput(scada_df=scada, test_wtg="T01", upgrade_timing=schedule)) + + assert out.p50_overall == pytest.approx(0.042) + toggle_df = captured["from_cfg_kwargs"]["toggle_df"] + assert sorted(toggle_df.columns) == ["toggle_off", "toggle_on"] + assert toggle_df["toggle_on"].any() + + +class TestExtractP50: + def test_picks_non_ref_test_row(self) -> None: + tdf = pd.DataFrame( + { + "test_wtg": ["T01", "T02", "T03"], + "p50_uplift": [0.05, 0.001, -0.002], + "is_ref": [False, True, True], + } + ) + assert _extract_p50(tdf, "T01") == pytest.approx(0.05) + + def test_raises_when_test_row_absent(self) -> None: + tdf = pd.DataFrame({"test_wtg": ["T02"], "p50_uplift": [0.0], "is_ref": [True]}) + with pytest.raises(ValueError, match="T01"): + _extract_p50(tdf, "T01") + + +def _stub_pipeline(monkeypatch) -> dict: # noqa: ANN001 + """Patch AssessmentInputs/run/combine to no-ops; return the captured-call dict.""" + captured: dict[str, object] = {} + + class FakeInputs: + @staticmethod + def from_cfg(**kwargs: object) -> str: + captured["from_cfg_kwargs"] = kwargs + return "inputs-sentinel" + + def fake_run(inputs: object) -> str: + captured["run_inputs"] = inputs + return "trdf-sentinel" + + def fake_combine(trdf: object, **kwargs: object) -> pd.DataFrame: + captured["combine_trdf"] = trdf + captured["combine_kwargs"] = kwargs + return pd.DataFrame({"test_wtg": ["T01"], "p50_uplift": [0.042], "is_ref": [False]}) + + monkeypatch.setattr(v0_binned, "AssessmentInputs", FakeInputs) + monkeypatch.setattr(v0_binned, "run_wind_up_analysis", fake_run) + monkeypatch.setattr(v0_binned, "combine_results", fake_combine) + return captured + + +class TestEstimateWiring: + def test_wires_pipeline_and_returns_p50(self, tmp_path, monkeypatch) -> None: # noqa: ANN001 + captured = _stub_pipeline(monkeypatch) + + method = V0BinnedMethod(_context(), scratch_dir=tmp_path) + out = method.estimate(_method_input(["T01", "T02", "T03", "T04"], "T01")) + + assert isinstance(out, MethodOutput) + assert out.p50_overall == pytest.approx(0.042) + assert out.p50_by_condition is None + assert captured["run_inputs"] == "inputs-sentinel" + assert captured["combine_trdf"] == "trdf-sentinel" + assert captured["combine_kwargs"]["auto_choose_refs"] is False + # the source-native scada is converted to wind-up format on the way into the pipeline, + # and the metadata/reanalysis from the context are handed through unchanged + kwargs = captured["from_cfg_kwargs"] + expected_scada = long_to_wind_up_format(_method_input(["T01", "T02", "T03", "T04"], "T01").scada_df) + assert kwargs["scada_df"].equals(expected_scada) + assert kwargs["reanalysis_datasets"] == [] + + def test_save_plots_defaults_off(self, tmp_path, monkeypatch) -> None: # noqa: ANN001 + captured = _stub_pipeline(monkeypatch) + method = V0BinnedMethod(_context(), scratch_dir=tmp_path) + method.estimate(_method_input(["T01", "T02", "T03", "T04"], "T01")) + assert captured["from_cfg_kwargs"]["plot_cfg"].save_plots is False + + def test_save_plots_on_writes_under_out_dir(self, tmp_path, monkeypatch) -> None: # noqa: ANN001 + captured = _stub_pipeline(monkeypatch) + method = V0BinnedMethod(_context(), scratch_dir=tmp_path, save_plots=True) + method.estimate(_method_input(["T01", "T02", "T03", "T04"], "T01")) + plot_cfg = captured["from_cfg_kwargs"]["plot_cfg"] + assert plot_cfg.save_plots is True + assert plot_cfg.plots_dir.parent == tmp_path / "v0_T01_20180101_20180701" + assert plot_cfg.plots_dir.name == "plots" diff --git a/tests/benchmarking/baselines/test_v0_end_to_end.py b/tests/benchmarking/baselines/test_v0_end_to_end.py new file mode 100644 index 00000000..0fc7283a --- /dev/null +++ b/tests/benchmarking/baselines/test_v0_end_to_end.py @@ -0,0 +1,66 @@ +"""End-to-end test: a real v0 run on a HoT-derived synthetic constant-Cp dataset. + +Downloads the Hill of Towie v2 SCADA (Zenodo) and ERA5 reanalysis (Open-Meteo), injects a +known constant-Cp uplift, runs the full wind_up pre/post analysis through ``V0BinnedMethod`` +behind the harness, and checks the recovered P50 lands near the injected truth. This is the +sanity that the v0 stack (incl. the source-native -> wind-up-format on-ramp) composes end to end. + +A real wind_up run per campaign is very slow, so this test is **opt-in**: it is skipped unless +``RUN_V0_E2E`` is set in the environment (e.g. ``RUN_V0_E2E=1 uv run pytest ``). It is +also marked ``slow`` so it never runs in the ``-m "not slow"`` fast gate even when opted in. +""" + +from __future__ import annotations + +import os + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.baselines.hot_context import build_hot_v0_context +from benchmarking.baselines.v0_binned import V0BinnedMethod +from benchmarking.harness import StudyConfig, score_study +from benchmarking.harness.example_hot_study import OracleMethod +from benchmarking.synthetic import ConstantCpChange +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + + +@pytest.mark.slow +@pytest.mark.skipif(not os.environ.get("RUN_V0_E2E"), reason="v0 e2e is very slow; set RUN_V0_E2E=1 to run it") +def test_v0_recovers_constant_cp_uplift(tmp_path) -> None: # noqa: ANN001 + # A lighter window than the driver (three Zenodo year zips): a 24-month baseline before an + # early-2018 upgrade plus a 6-month campaign, enough to exercise the full stack on one campaign. + scada_df, _metadata_df = load_hot_scada( + start_dt=pd.Timestamp("2016-01-01", tz="UTC"), + end_dt_excl=pd.Timestamp("2018-09-01", tz="UTC"), + wtg_numbers=[1, 3, 4, 7], + ) + context = build_hot_v0_context(wtg_names=["T01", "T03", "T04", "T07"]) + study = StudyConfig( + mode="prepost", + turbine_subset=["T01", "T03", "T04", "T07"], + treatment_start_range=(pd.Timestamp("2018-02-01", tz="UTC"), pd.Timestamp("2018-02-08", tz="UTC")), + min_pre_months=24, + campaign_months=[6], + n_replicates=1, + seed=0, + ) + + results = score_study( + scada_df, + profile=[ConstantCpChange(delta=0.05)], + methods=[V0BinnedMethod(context, scratch_dir=tmp_path), OracleMethod(scada_df)], + study=study, + profile_name="constant_cp", + ) + + oracle_row = results.loc[results["method"] == "oracle"].iloc[0] + v0_row = results.loc[results["method"] == "v0_binned"].iloc[0] + + # the harness is wired correctly: the oracle recovers the injected truth exactly + assert abs(oracle_row["signed_error"]) < 1e-6 + # a real constant-Cp uplift is positive, and v0 recovers it to within a few percentage points + assert v0_row["truth"] > 0 + assert np.isfinite(v0_row["estimate"]) + assert abs(v0_row["signed_error"]) < 0.03 diff --git a/tests/benchmarking/diagnostics/__init__.py b/tests/benchmarking/diagnostics/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/benchmarking/diagnostics/test_diagnostics.py b/tests/benchmarking/diagnostics/test_diagnostics.py new file mode 100644 index 00000000..515f70fe --- /dev/null +++ b/tests/benchmarking/diagnostics/test_diagnostics.py @@ -0,0 +1,219 @@ +"""Smoke tests for the shared benchmarking diagnostics package. + +These assert the plotting/config functions run clean (the suite treats warnings as errors) and +write the expected files on a tiny synthetic frame, and that plots needing an absent signal skip +gracefully. They do not assert pixel content — image fidelity is reviewed by eye. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib.pyplot as plt +import numpy as np +import pandas as pd +import pytest +import yaml + +from benchmarking.diagnostics import write_common_diagnostics, write_run_config +from benchmarking.diagnostics.context import ERA5_WD_COL, ERA5_WS_COL, DiagnosticContext +from benchmarking.diagnostics.coverage import exclusion_bucket, plot_excluded_fraction +from benchmarking.diagnostics.density import density_scatter +from benchmarking.synthetic import ColumnSchema + +if TYPE_CHECKING: + from pathlib import Path + +_FULL_COLUMNS = ColumnSchema( + turbine="turbine", + active_power="power", + wind_speed="ws", + wind_speed_sd="ws_sd", + gen_rpm="rpm", + pitch="pitch", + reactive_power="reactive", + nacelle_position="nacelle", + ambient_temp="temp", + availability="avail", +) + + +def _long_scada(index: pd.DatetimeIndex, turbines: list[str], *, rng: np.random.Generator) -> pd.DataFrame: + """A small long-format SCADA frame with every diagnostic signal, indexed by timestamp.""" + frames = [] + for turbine in turbines: + ws = rng.uniform(3, 18, len(index)) + power = np.clip(0.5 * ws**3, 0, 2300) + rng.normal(0, 20, len(index)) + frames.append( + pd.DataFrame( + { + "turbine": turbine, + "power": power, + "ws": ws, + "ws_sd": rng.uniform(0.3, 2.0, len(index)), + "rpm": np.clip(ws * 90, 0, 1600), + "pitch": rng.uniform(-2, 25, len(index)), + "reactive": rng.normal(0, 50, len(index)), + "nacelle": rng.uniform(0, 360, len(index)), + "temp": rng.uniform(-5, 25, len(index)), + "avail": 600.0, + }, + index=index, + ) + ) + return pd.concat(frames) + + +def _context( + tmp_path: Path, + *, + columns: ColumnSchema = _FULL_COLUMNS, + with_era5: bool = True, + excluded: np.ndarray | None = None, +) -> DiagnosticContext: + rng = np.random.default_rng(0) + index = pd.date_range("2020-01-01", periods=300, freq="10min", tz="UTC") + scada = _long_scada(index, ["T1", "T2", "T3"], rng=rng) + treated = np.asarray(index >= index[len(index) // 2]) + used = rng.random(len(index)) > 0.1 + era5 = None + if with_era5: + era5 = pd.DataFrame( + {ERA5_WS_COL: rng.uniform(3, 18, len(index)), ERA5_WD_COL: rng.uniform(0, 360, len(index))}, index=index + ) + return DiagnosticContext( + run_dir=tmp_path / "run", + test_wtg="T1", + turbine_col="turbine", + columns=columns, + scada_df=scada, + treated_ts=treated, + used_ts=used, + timebase=pd.Timedelta(minutes=10), + mode="prepost", + era5_df=era5, + excluded_ts=excluded, + ) + + +# --- caller-flagged row exclusions -------------------------------------------------------------- + + +def _excluded_mask(n: int = 300, *, every: int = 5) -> np.ndarray: + return np.arange(n) % every == 0 + + +class TestExclusionTimelineBucket: + """A single averaged dot is not a timeline: the bucket has to suit the campaign's length.""" + + def test_short_campaign_buckets_daily(self) -> None: + index = pd.date_range("2020-01-01", periods=300, freq="10min", tz="UTC") + assert exclusion_bucket(index) == "1D" + + def test_long_campaign_buckets_weekly(self) -> None: + index = pd.date_range("2020-01-01", periods=3 * 365 * 24, freq="h", tz="UTC") + assert exclusion_bucket(index) == "7D" + + def test_a_short_campaign_gets_more_than_one_point(self, tmp_path: Path) -> None: + ctx = _context(tmp_path, excluded=_excluded_mask()) # the fixture spans ~2 days + line = plot_excluded_fraction(ctx) + assert line is not None + assert len(pd.Series(_excluded_mask(), index=ctx.index).resample(exclusion_bucket(ctx.index)).mean()) > 1 + + +class TestExcludedRowPlots: + """The exclusion view is the *usual* 2x3 operating-curve figure, coloured kept vs excluded.""" + + def test_written_when_rows_are_excluded(self, tmp_path: Path) -> None: + ctx = _context(tmp_path, excluded=_excluded_mask()) + names = {p.name for p in write_common_diagnostics(ctx)} + assert "ops_curves_excluded.png" in names + assert "excluded_row_fraction.png" in names + + def test_lands_in_the_filter_stage_folder(self, tmp_path: Path) -> None: + ctx = _context(tmp_path, excluded=_excluded_mask()) + written = {p.name: p for p in write_common_diagnostics(ctx)} + assert written["ops_curves_excluded.png"].parent.name == written["filter_coverage.png"].parent.name + + def test_skipped_when_the_method_excludes_nothing(self, tmp_path: Path) -> None: + """A clean campaign must not sprout an empty plot.""" + ctx = _context(tmp_path, excluded=np.zeros(300, dtype=bool)) + names = {p.name for p in write_common_diagnostics(ctx)} + assert "ops_curves_excluded.png" not in names + assert "excluded_row_fraction.png" not in names + + def test_skipped_when_the_method_has_no_exclusion_concept(self, tmp_path: Path) -> None: + ctx = _context(tmp_path) + assert ctx.excluded_ts is None + names = {p.name for p in write_common_diagnostics(ctx)} + assert "ops_curves_excluded.png" not in names + + def test_a_misaligned_exclusion_mask_skips_every_diagnostic(self, tmp_path: Path) -> None: + """Same contract as the other masks: a length mismatch is a caller bug, not a partial plot.""" + ctx = _context(tmp_path, excluded=np.zeros(7, dtype=bool)) + assert write_common_diagnostics(ctx) == [] + + +def test_common_diagnostics_writes_expected_plots(tmp_path: Path) -> None: + ctx = _context(tmp_path) + written = write_common_diagnostics(ctx) + names = {p.name for p in written} + expected = { + "input_data_timeline.png", + "input_data_coverage.png", + "filter_coverage.png", + "condition_histograms.png", + "ops_curves.png", + "ops_curves_kept_only.png", + "ops_curves_by_upgrade.png", + "reactive_vs_active.png", + "power_factor.png", + "northing_error.png", + } + assert expected <= names + assert all(p.exists() for p in written) + + +def test_optional_signals_skip_gracefully(tmp_path: Path) -> None: + bare = ColumnSchema( + turbine="turbine", + active_power="power", + wind_speed="ws", + wind_speed_sd="ws_sd", + gen_rpm="rpm", + availability="avail", + ) + ctx = _context(tmp_path, columns=bare, with_era5=False) + written = write_common_diagnostics(ctx) + names = {p.name for p in written} + # reactive needs a reactive tag; northing needs nacelle + ERA5 — both absent here. + assert "reactive_vs_active.png" not in names + assert "northing_error.png" not in names + # the core curves still render. + assert "ops_curves.png" in names + + +def test_write_run_config_yaml(tmp_path: Path) -> None: + ctx = _context(tmp_path) + path = write_run_config(ctx, method_name="unit_test", method_params={"foo": 1}, extra={"era5_lag_rows": 3}) + assert path.exists() + record = yaml.safe_load(path.read_text()) + assert record["method"] == "unit_test" + assert record["mode"] == "prepost" + assert record["test_wtg"] == "T1" + assert record["references"] == ["T2", "T3"] + assert record["extra"]["era5_lag_rows"] == 3 + + +def test_density_scatter_degenerate_input_does_not_raise() -> None: + _fig, ax = plt.subplots() + # all-identical x (no spread) must fall back to a flat colour rather than erroring. + density_scatter(np.ones(20), np.arange(20.0), ax=ax) + plt.close("all") + + +@pytest.mark.parametrize("with_era5", [True, False]) +def test_runs_without_era5(tmp_path: Path, *, with_era5: bool) -> None: + ctx = _context(tmp_path, with_era5=with_era5) + written = write_common_diagnostics(ctx) + assert ("northing_error.png" in {p.name for p in written}) == with_era5 diff --git a/tests/benchmarking/harness/__init__.py b/tests/benchmarking/harness/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/benchmarking/harness/stubs.py b/tests/benchmarking/harness/stubs.py new file mode 100644 index 00000000..0292b026 --- /dev/null +++ b/tests/benchmarking/harness/stubs.py @@ -0,0 +1,164 @@ +"""Stub methods for exercising the scoring machinery without a real estimator. + +The oracle computes the true uplift from its *own* method-input window (windowed synthetic vs +original), so if the harness ever scored truth over different records than it handed the +method, the oracle's error would stop being zero. Biased/Noisy wrap the oracle to drive the +bias/precision metrics; Recording captures inputs for the fairness test. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd + +from benchmarking.harness.conditions import CONDITION_BINS, condition_bins, energy_ratio_by_bin +from benchmarking.harness.method import MethodInput, MethodOutput +from benchmarking.synthetic import HOT_COLUMNS, treated_mask +from benchmarking.synthetic.sources.hill_of_towie import HOT_RATED_POWER_KW + + +def oracle_overall_uplift(mi: MethodInput, original_df: pd.DataFrame) -> float: + """Energy-ratio uplift over the treated test-turbine rows in ``mi``'s window.""" + syn = mi.scada_df + test_rows = syn[syn[mi.turbine_col] == mi.test_wtg] + treated = treated_mask(test_rows.index, mi.upgrade_timing) + treated_rows = test_rows[treated] + + syn_power = treated_rows[HOT_COLUMNS.active_power].to_numpy(dtype=float) + orig_test = original_df[original_df[mi.turbine_col] == mi.test_wtg] + orig_power = orig_test.loc[treated_rows.index, HOT_COLUMNS.active_power].to_numpy(dtype=float) + + finite = np.isfinite(syn_power) & np.isfinite(orig_power) + denom = orig_power[finite].sum() + return syn_power[finite].sum() / denom - 1.0 if denom else float("nan") + + +class OracleMethod: + """Returns the true uplift; signed error should be ~0.""" + + def __init__(self, original_df: pd.DataFrame, name: str = "oracle") -> None: + self._original = original_df + self.name = name + + def estimate(self, mi: MethodInput) -> MethodOutput: + return MethodOutput(p50_overall=oracle_overall_uplift(mi, self._original)) + + +class BiasedMethod: + """Returns the true uplift plus a fixed offset; signed error should equal the offset.""" + + def __init__(self, original_df: pd.DataFrame, offset: float, name: str = "biased") -> None: + self._original = original_df + self._offset = offset + self.name = name + + def estimate(self, mi: MethodInput) -> MethodOutput: + return MethodOutput(p50_overall=oracle_overall_uplift(mi, self._original) + self._offset) + + +class NoisyMethod: + """Returns the true uplift plus seeded Gaussian noise; error spread should track sigma.""" + + def __init__(self, original_df: pd.DataFrame, sigma: float, seed: int = 0, name: str = "noisy") -> None: + self._original = original_df + self._sigma = sigma + self._rng = np.random.default_rng(seed) + self.name = name + + def estimate(self, mi: MethodInput) -> MethodOutput: + noise = float(self._rng.normal(0.0, self._sigma)) + return MethodOutput(p50_overall=oracle_overall_uplift(mi, self._original) + noise) + + +class RecordingMethod: + """Captures every MethodInput it is handed, for the fairness test.""" + + def __init__(self, name: str = "recording") -> None: + self.name = name + self.seen: list[MethodInput] = [] + + def estimate(self, mi: MethodInput) -> MethodOutput: + captured = MethodInput( + scada_df=mi.scada_df.copy(), + test_wtg=mi.test_wtg, + upgrade_timing=mi.upgrade_timing, + turbine_col=mi.turbine_col, + ) + self.seen.append(captured) + return MethodOutput(p50_overall=0.0) + + +def conditional_oracle_by_condition(mi: MethodInput, original_df: pd.DataFrame) -> pd.DataFrame: + """Per-bin oracle uplift over treated rows, binned on the test turbine's measured ws/ti.""" + syn = mi.scada_df + test_rows = syn[syn[mi.turbine_col] == mi.test_wtg] + treated = treated_mask(test_rows.index, mi.upgrade_timing) + treated_rows = test_rows[treated] + orig_test = original_df[original_df[mi.turbine_col] == mi.test_wtg] + actual = treated_rows[HOT_COLUMNS.active_power].to_numpy(dtype=float) + counterfactual = orig_test.loc[treated_rows.index, HOT_COLUMNS.active_power].to_numpy(dtype=float) + ws = treated_rows[HOT_COLUMNS.wind_speed].to_numpy(dtype=float) + sd = treated_rows[HOT_COLUMNS.wind_speed_sd].to_numpy(dtype=float) + ti = np.divide(sd, ws, out=np.full_like(sd, np.nan), where=ws != 0) + # power bins on the untreated operating point; for the oracle the counterfactual IS the original + # power, so binning on it matches the truth's original-power binning (near-zero error end-to-end). + axes = ( + ("ws", ws, CONDITION_BINS["ws"]), + ("ti", ti, CONDITION_BINS["ti"]), + ("power", counterfactual, condition_bins("power", rated_power_kw=HOT_RATED_POWER_KW)), + ) + frames = [] + for name, values, bins in axes: + table = energy_ratio_by_bin(values, actual, counterfactual, bins=bins) + table.insert(0, "condition", name) + frames.append(table[["condition", "condition_bin", "p50_uplift"]]) + return pd.concat(frames, ignore_index=True) + + +class ConditionalOracleMethod: + """Oracle that also emits per-bin oracle uplift; conditional signed error should be ~0.""" + + def __init__(self, original_df: pd.DataFrame, name: str = "cond_oracle") -> None: + self._original = original_df + self.name = name + + def estimate(self, mi: MethodInput) -> MethodOutput: + return MethodOutput( + p50_overall=oracle_overall_uplift(mi, self._original), + p50_by_condition=conditional_oracle_by_condition(mi, self._original), + ) + + +class UncertainMethod: + """Reports a fixed sigma and diagnostics, for exercising the seam's uncertainty passthrough. + + Deliberately dumb: the numbers mean nothing, they only have to arrive intact and in the right + place. ``diagnostic_columns`` lets a test drive the clash guard in ``_merge_diagnostics``. + """ + + def __init__( + self, + original_df: pd.DataFrame, + sigma: float = 0.01, + name: str = "uncertain", + diagnostic_columns: dict[str, float] | None = None, + ) -> None: + self._original = original_df + self._sigma = sigma + self._diagnostics = {"n_blocks": 7.0} if diagnostic_columns is None else diagnostic_columns + self.name = name + + def estimate(self, mi: MethodInput) -> MethodOutput: + by_condition = conditional_oracle_by_condition(mi, self._original) + by_condition["sigma_uplift"] = self._sigma * 2.0 + rows = [{"condition": "overall", "condition_bin": "overall", **self._diagnostics}] + rows += [ + {"condition": c, "condition_bin": b, **self._diagnostics} + for c, b in zip(by_condition["condition"], by_condition["condition_bin"], strict=True) + ] + return MethodOutput( + p50_overall=oracle_overall_uplift(mi, self._original), + p50_by_condition=by_condition, + sigma_overall=self._sigma, + uncertainty_diagnostics=pd.DataFrame(rows), + ) diff --git a/tests/benchmarking/harness/test_calibration.py b/tests/benchmarking/harness/test_calibration.py new file mode 100644 index 00000000..e2ef095d --- /dev/null +++ b/tests/benchmarking/harness/test_calibration.py @@ -0,0 +1,132 @@ +"""Tests for the uncertainty-calibration metrics. + +Built on constructed z's with known properties, so a target the metric should hit is known exactly +rather than inferred from a method's behaviour. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.harness.calibration import ( + TARGET_COVERAGE_1SIGMA, + calibration_summary, + coverage_standard_error, + summarize_calibration, +) + + +def _errors_from_z(z: np.ndarray, sigma: float = 0.01) -> tuple[np.ndarray, np.ndarray]: + """Signed errors and sigmas whose ratio is exactly ``z``.""" + return z * sigma, np.full(len(z), sigma) + + +class TestCalibrationSummary: + def test_a_perfectly_calibrated_sigma_hits_the_target(self) -> None: + z = np.random.default_rng(0).standard_normal(20000) + summary = calibration_summary(*_errors_from_z(z)) + assert summary.coverage_1sigma == pytest.approx(TARGET_COVERAGE_1SIGMA, abs=0.01) + assert summary.z_spread == pytest.approx(1.0, abs=0.02) + assert summary.z_robust == pytest.approx(1.0, abs=0.03) + + def test_a_sigma_that_is_too_small_under_covers(self) -> None: + z = np.random.default_rng(0).standard_normal(20000) * 2.0 # errors twice as wide as claimed + summary = calibration_summary(*_errors_from_z(z)) + assert summary.coverage_1sigma < 0.45 + assert summary.z_spread == pytest.approx(2.0, abs=0.05) + + def test_a_sigma_that_is_too_large_over_covers_and_shows_the_width(self) -> None: + z = np.random.default_rng(0).standard_normal(20000) * 0.5 + summary = calibration_summary(*_errors_from_z(z, sigma=0.02)) + assert summary.coverage_1sigma > 0.9 + # Over-covering is bought with width: mean_sigma is what exposes it. + assert summary.mean_sigma == pytest.approx(0.02) + assert summary.rms_error == pytest.approx(0.01, rel=0.05) + + def test_bias_alone_can_break_coverage(self) -> None: + """A method with no scatter but a 2-sigma bias covers nothing, which is the honest read.""" + errors = np.full(500, 0.02) + sigma = np.full(500, 0.01) + summary = calibration_summary(errors, sigma) + assert summary.coverage_1sigma == 0.0 + assert summary.z_spread == pytest.approx(0.0) + + def test_z_spread_blows_up_on_a_tail_while_z_robust_holds(self) -> None: + """The pair's whole purpose: disagreement localises the miscalibration to the tails.""" + z = np.random.default_rng(0).standard_normal(2000) + z[0] = 300.0 + summary = calibration_summary(*_errors_from_z(z)) + assert summary.z_spread > 5 + assert summary.z_robust == pytest.approx(1.0, abs=0.1) + + +class TestUnusable: + def test_a_non_positive_sigma_is_excluded_but_counted(self) -> None: + errors = np.array([0.01, 0.01, 0.01, 0.01]) + sigma = np.array([0.01, 0.0, -1.0, np.nan]) + summary = calibration_summary(errors, sigma) + assert summary.n == 1 + assert summary.n_unusable == 3 + + def test_a_non_finite_error_is_ignored_entirely(self) -> None: + """No estimate means nothing to calibrate: it is not an uncertainty failure.""" + summary = calibration_summary(np.array([0.01, np.nan]), np.array([0.01, np.nan])) + assert summary.n == 1 + assert summary.n_unusable == 0 + + def test_all_unusable_gives_a_nan_summary(self) -> None: + summary = calibration_summary(np.array([0.01, 0.02]), np.array([np.nan, 0.0])) + assert summary.n == 0 + assert summary.n_unusable == 2 + assert np.isnan(summary.coverage_1sigma) + + def test_empty_input_gives_a_nan_summary(self) -> None: + summary = calibration_summary(np.array([]), np.array([])) + assert summary.n == 0 + assert np.isnan(summary.z_spread) + + def test_mismatched_lengths_raise(self) -> None: + with pytest.raises(ValueError, match="same length"): + calibration_summary(np.array([0.01]), np.array([0.01, 0.02])) + + +class TestSummarizeCalibration: + def _frame(self) -> pd.DataFrame: + rng = np.random.default_rng(0) + good = rng.standard_normal(4000) + bad = rng.standard_normal(4000) * 3.0 + return pd.DataFrame( + { + "group": ["good"] * 4000 + ["bad"] * 4000, + "signed_error": np.concatenate([good, bad]) * 0.01, + "sigma": 0.01, + } + ) + + def test_groups_are_summarised_independently(self) -> None: + table = summarize_calibration(self._frame(), group_keys=["group"]).set_index("group") + assert table.loc["good", "coverage_1sigma"] == pytest.approx(TARGET_COVERAGE_1SIGMA, abs=0.03) + assert table.loc["bad", "coverage_1sigma"] < 0.35 + + def test_missing_columns_raise(self) -> None: + with pytest.raises(ValueError, match="missing column"): + summarize_calibration(pd.DataFrame({"group": ["a"]}), group_keys=["group"]) + + def test_multiple_group_keys(self) -> None: + frame = self._frame().assign(other=lambda d: np.where(d.index < 2000, "x", "y")) + table = summarize_calibration(frame, group_keys=["group", "other"]) + assert len(table) == 3 # good/x, good/y, bad/y + assert list(table.columns[:2]) == ["group", "other"] + + +class TestCoverageStandardError: + def test_shrinks_as_the_root_of_n(self) -> None: + assert coverage_standard_error(64) == pytest.approx(coverage_standard_error(16) / 2, rel=1e-6) + + def test_at_sixty_four_independent_draws(self) -> None: + assert coverage_standard_error(64) == pytest.approx(0.058, abs=0.001) + + def test_non_positive_n_is_nan(self) -> None: + assert np.isnan(coverage_standard_error(0)) diff --git a/tests/benchmarking/harness/test_campaign.py b/tests/benchmarking/harness/test_campaign.py new file mode 100644 index 00000000..8229d517 --- /dev/null +++ b/tests/benchmarking/harness/test_campaign.py @@ -0,0 +1,159 @@ +"""Tests for campaign windows and their two derived selections.""" + +from __future__ import annotations + +from itertools import pairwise + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.harness.campaign import ( + CampaignWindow, + campaign_windows, + treated_activity_mask, + window_row_mask, +) +from benchmarking.synthetic import ToggleSchedule + +T0 = pd.Timestamp("2018-01-01", tz="UTC") + + +def test_window_bounds_use_fixed_baseline_and_activity_length() -> None: + [window] = campaign_windows(T0, min_pre_months=12, campaign_months=[6]) + assert isinstance(window, CampaignWindow) + assert window.treatment_start == T0 + assert window.baseline_start == T0 - pd.DateOffset(months=12) + assert window.activity_end == T0 + pd.DateOffset(months=6) + assert window.months == 6 + + +def test_one_window_per_campaign_length() -> None: + windows = campaign_windows(T0, min_pre_months=12, campaign_months=[3, 6, 9, 12]) + assert [w.months for w in windows] == [3, 6, 9, 12] + + +def test_shorter_windows_share_baseline_and_treatment_start() -> None: + windows = campaign_windows(T0, min_pre_months=12, campaign_months=[3, 6, 9, 12]) + assert {w.baseline_start for w in windows} == {T0 - pd.DateOffset(months=12)} + assert {w.treatment_start for w in windows} == {T0} + activity_ends = [w.activity_end for w in windows] + assert activity_ends == sorted(activity_ends) # strictly growing post window + + +def test_each_windows_row_mask_is_a_prefix_subset_of_the_next() -> None: + index = pd.date_range("2017-01-01", "2019-01-01", freq="6h", tz="UTC") + windows = campaign_windows(T0, min_pre_months=12, campaign_months=[3, 6, 9, 12]) + masks = [window_row_mask(index, w) for w in windows] + for shorter, longer in pairwise(masks): + # 3 ⊂ 6 ⊂ 9 ⊂ 12: every row in the shorter window is in the longer one + assert np.all(longer[shorter]) + assert longer.sum() > shorter.sum() + + +def test_window_row_mask_selects_baseline_through_activity_end() -> None: + index = pd.date_range("2017-01-01", "2019-01-01", freq="1D", tz="UTC") + [window] = campaign_windows(T0, min_pre_months=12, campaign_months=[6]) + mask = window_row_mask(index, window) + selected = index[mask] + assert selected.min() >= window.baseline_start + assert selected.max() < window.activity_end + unselected = index[~mask] + assert ((unselected < window.baseline_start) | (unselected >= window.activity_end)).all() + + +def test_prepost_truth_mask_is_the_post_rows_within_activity() -> None: + index = pd.date_range("2017-01-01", "2019-06-01", freq="1D", tz="UTC") + [window] = campaign_windows(T0, min_pre_months=12, campaign_months=[6]) + mask = treated_activity_mask(index, T0, window=window) # prepost: a bare timestamp + selected = index[mask] + assert (selected >= T0).all() + assert (selected < window.activity_end).all() + assert not mask[index < T0].any() # baseline never counted as treated + + +def test_toggle_truth_mask_excludes_baseline_and_keeps_only_on_rows() -> None: + index = pd.date_range("2017-01-01", "2019-06-01", freq="6h", tz="UTC") + schedule = ToggleSchedule(period=pd.Timedelta(days=14), start=T0) + [window] = campaign_windows(T0, min_pre_months=12, campaign_months=[6]) + mask = treated_activity_mask(index, schedule, window=window) + assert not mask[index < T0].any() # baseline untreated + assert mask.any() # some on-rows inside the toggling window + assert (index[mask] < window.activity_end).all() + assert mask.sum() < ((index >= T0) & (index < window.activity_end)).sum() # only on-blocks + + +def test_infeasible_lengths_are_dropped_when_data_bounds_given() -> None: + # data ends only 4 months after t0, so 6/9/12-month campaigns do not fit + windows = campaign_windows( + T0, + min_pre_months=12, + campaign_months=[3, 6, 9, 12], + data_start=pd.Timestamp("2016-01-01", tz="UTC"), + data_end=T0 + pd.DateOffset(months=4), + ) + assert [w.months for w in windows] == [3] + + +class TestCampaignWeeks: + """A weeks grid: the same window machinery on a ``DateOffset(weeks=...)`` activity length.""" + + def test_window_bounds_use_week_activity_length(self) -> None: + [window] = campaign_windows(T0, min_pre_months=12, campaign_weeks=[2]) + assert window.treatment_start == T0 + assert window.baseline_start == T0 - pd.DateOffset(months=12) # baseline stays months-based + assert window.activity_end == T0 + pd.DateOffset(weeks=2) + assert window.length == 2 + assert window.unit == "weeks" + assert window.length_col == "campaign_weeks" + + def test_one_window_per_campaign_length(self) -> None: + windows = campaign_windows(T0, min_pre_months=12, campaign_weeks=[1, 2, 4, 8]) + assert [w.length for w in windows] == [1, 2, 4, 8] + + def test_shorter_windows_are_prefixes_of_longer_ones(self) -> None: + index = pd.date_range("2016-06-01", "2018-06-01", freq="6h", tz="UTC") + windows = campaign_windows(T0, min_pre_months=12, campaign_weeks=[1, 2, 4, 8]) + masks = [window_row_mask(index, w) for w in windows] + for shorter, longer in pairwise(masks): + assert np.all(longer[shorter]) + assert longer.sum() > shorter.sum() + + def test_infeasible_lengths_are_dropped(self) -> None: + windows = campaign_windows( + T0, + min_pre_months=12, + campaign_weeks=[1, 2, 4, 8], + data_start=pd.Timestamp("2016-01-01", tz="UTC"), + data_end=T0 + pd.DateOffset(weeks=3), + ) + assert [w.length for w in windows] == [1, 2] + + def test_months_property_raises_on_a_weeks_window(self) -> None: + # the months-only inspect_* scripts read ``window.months``; a weeks window must fail loudly + # there rather than silently reporting a week count as a month count. + [window] = campaign_windows(T0, min_pre_months=12, campaign_weeks=[2]) + with pytest.raises(ValueError, match="unit='weeks'"): + _ = window.months + + +class TestCampaignLengthValidation: + """Exactly one of the two grids must be given.""" + + def test_neither_grid_raises(self) -> None: + with pytest.raises(ValueError, match="exactly one"): + campaign_windows(T0, min_pre_months=12) + + def test_both_grids_raise(self) -> None: + with pytest.raises(ValueError, match="exactly one"): + campaign_windows(T0, min_pre_months=12, campaign_months=[3], campaign_weeks=[2]) + + +def test_months_window_still_reports_months_unit() -> None: + # back-compat: the months path is unchanged, so ``months`` keeps working and the length column + # keeps its existing name. + [window] = campaign_windows(T0, min_pre_months=12, campaign_months=[6]) + assert window.unit == "months" + assert window.length == 6 + assert window.months == 6 + assert window.length_col == "campaign_months" diff --git a/tests/benchmarking/harness/test_conditions.py b/tests/benchmarking/harness/test_conditions.py new file mode 100644 index 00000000..2dfe139b --- /dev/null +++ b/tests/benchmarking/harness/test_conditions.py @@ -0,0 +1,117 @@ +"""Tests for the shared condition bins and the binned energy-ratio reducer.""" + +from __future__ import annotations + +import numpy as np +import pytest + +from benchmarking.harness.conditions import ( + CONDITION_BINS, + CONDITIONS, + POWER_FRACTION_EDGES, + TI_BINS, + WS_BINS, + condition_bins, + energy_ratio_by_bin, + validate_conditions, +) + + +class TestValidateConditions: + """The shared check each method runs on its caller-supplied ``conditions``.""" + + def test_supported_conditions_pass(self) -> None: + validate_conditions(("ws", "power"), supported=("ws", "ti", "power"), method_name="power_model") + + def test_empty_conditions_pass(self) -> None: + # opting out of conditional reporting entirely is legitimate + validate_conditions((), supported=("power",), method_name="toggle_specialist") + + def test_unknown_condition_raises(self) -> None: + with pytest.raises(ValueError, match="unknown condition"): + validate_conditions(("bogus",), supported=("ws", "ti", "power"), method_name="power_model") + + def test_known_but_unsupported_condition_raises(self) -> None: + # "ws" is a real axis, just not one toggle_specialist can offer -- a different error from a typo + with pytest.raises(ValueError, match="does not support"): + validate_conditions(("ws",), supported=("power",), method_name="toggle_specialist") + + def test_unsupported_error_names_the_method_and_what_it_supports(self) -> None: + with pytest.raises(ValueError, match="toggle_specialist") as excinfo: + validate_conditions(("ti",), supported=("power",), method_name="toggle_specialist") + assert "power" in str(excinfo.value) + + +def test_bin_edges_have_expected_width() -> None: + assert WS_BINS[0] == 0.0 + assert WS_BINS[-1] == 26.0 + assert np.allclose(np.diff(WS_BINS), 2.0) + assert TI_BINS[0] == 0.0 + assert TI_BINS[-1] == 0.5 + assert np.allclose(np.diff(TI_BINS), 0.05) + assert CONDITIONS == ("ws", "ti", "power") + assert CONDITION_BINS == {"ws": WS_BINS, "ti": TI_BINS} + + +def test_power_fraction_edges_center_bins_on_round_fractions() -> None: + # 6 bins whose midpoints are 0, 0.2, ..., 1.0 of rated; outer edges pushed just beyond + # [0, rated] so pd.cut keeps slightly-negative (cut-in) and slightly-over-rated (noise) power. + assert POWER_FRACTION_EDGES == [-0.1, 0.1, 0.3, 0.5, 0.7, 0.9, 1.1] + midpoints = (np.array(POWER_FRACTION_EDGES[:-1]) + np.array(POWER_FRACTION_EDGES[1:])) / 2 + assert np.allclose(midpoints, [0.0, 0.2, 0.4, 0.6, 0.8, 1.0]) + + +def test_condition_bins_returns_fixed_edges_for_ws_and_ti() -> None: + # ws/ti are treatment-invariant fixed edges; the rating is accepted but ignored. + assert condition_bins("ws", rated_power_kw=2300.0) == WS_BINS + assert condition_bins("ti", rated_power_kw=2300.0) == TI_BINS + + +def test_condition_bins_scales_power_edges_by_rating() -> None: + assert condition_bins("power", rated_power_kw=2300.0) == [-230.0, 230.0, 690.0, 1150.0, 1610.0, 2070.0, 2530.0] + + +def test_condition_bins_requires_rating_for_power() -> None: + with pytest.raises(ValueError, match="rated_power_kw"): + condition_bins("power") + + +def test_condition_bins_rejects_unknown_condition() -> None: + with pytest.raises(ValueError, match="unknown condition"): + condition_bins("gustiness", rated_power_kw=2300.0) + + +def test_energy_ratio_by_bin_computes_per_bin_ratio() -> None: + # two rows in (4,6], two in (6,8]; counterfactual constant so uplift = mean(actual)/cf - 1 + cond = np.array([5.0, 5.0, 7.0, 7.0]) + actual = np.array([110.0, 90.0, 150.0, 150.0]) + counterfactual = np.array([100.0, 100.0, 100.0, 100.0]) + out = energy_ratio_by_bin(cond, actual, counterfactual, bins=WS_BINS) + by = out.set_index("condition_bin") + assert by.loc["(4.0, 6.0]", "p50_uplift"] == 0.0 # (110+90)/200 - 1 + assert by.loc["(6.0, 8.0]", "p50_uplift"] == 0.5 # 300/200 - 1 + assert by.loc["(4.0, 6.0]", "n_records"] == 2 + + +def test_energy_ratio_by_bin_is_nan_safe_and_covers_all_bins() -> None: + cond = np.array([5.0, np.nan, 7.0]) + actual = np.array([100.0, 50.0, np.nan]) + counterfactual = np.array([100.0, 50.0, 100.0]) + out = energy_ratio_by_bin(cond, actual, counterfactual, bins=WS_BINS) + # one row per bin edge interval, no warnings, empty bins -> NaN uplift + assert len(out) == len(WS_BINS) - 1 + assert out["condition_bin"].is_unique + assert out.loc[out["condition_bin"] == "(6.0, 8.0]", "p50_uplift"].isna().all() + + +def test_energy_ratio_by_bin_exposes_per_bin_sums() -> None: + # the per-bin actual/counterfactual energy sums are needed to re-level a decomposition to an overall + cond = np.array([5.0, 5.0, 7.0]) + actual = np.array([110.0, 90.0, 150.0]) + counterfactual = np.array([100.0, 100.0, 100.0]) + by = energy_ratio_by_bin(cond, actual, counterfactual, bins=WS_BINS).set_index("condition_bin") + assert by.loc["(4.0, 6.0]", "sum_actual"] == 200.0 + assert by.loc["(4.0, 6.0]", "sum_counterfactual"] == 200.0 + assert by.loc["(6.0, 8.0]", "sum_actual"] == 150.0 + # empty bins carry a zero energy sum (they contribute nothing to an aggregation) + assert by.loc["(8.0, 10.0]", "sum_actual"] == 0.0 diff --git a/tests/benchmarking/harness/test_hot_end_to_end.py b/tests/benchmarking/harness/test_hot_end_to_end.py new file mode 100644 index 00000000..ce072f2d --- /dev/null +++ b/tests/benchmarking/harness/test_hot_end_to_end.py @@ -0,0 +1,59 @@ +"""Slow end-to-end test: a real Hill of Towie constant-Cp study through the full harness. + +Network- and data-heavy (downloads HoT SCADA from Zenodo on first run); excluded from the +default offline suite via the ``slow`` marker. Proves the harness runs a real study end to +end: load -> inject -> replicate ensemble -> campaign sweep -> score -> leaderboard -> plot. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.harness import StudyConfig, leaderboard, plot_campaign_curves, score_study +from benchmarking.synthetic import ConstantCpChange +from benchmarking.synthetic.sources.hill_of_towie import load_hot_scada + +from .stubs import BiasedMethod, OracleMethod + +if TYPE_CHECKING: + from pathlib import Path + +pytestmark = pytest.mark.slow + + +def test_constant_cp_study_on_real_hot_data(tmp_path: Path) -> None: + scada_df, _ = load_hot_scada( + start_dt=pd.Timestamp("2016-01-01", tz="UTC"), + end_dt_excl=pd.Timestamp("2017-05-01", tz="UTC"), + wtg_numbers=[1, 3, 4, 7], # the stable SW turbines T01, T03, T04, T07 + ) + + study = StudyConfig( + mode="prepost", + turbine_subset=["T01", "T03", "T04", "T07"], + treatment_start_range=(pd.Timestamp("2017-01-01", tz="UTC"), pd.Timestamp("2017-01-31", tz="UTC")), + min_pre_months=12, + campaign_months=[3], + n_replicates=2, + seed=0, + ) + methods = [OracleMethod(scada_df), BiasedMethod(scada_df, offset=0.02)] + + results = score_study( + scada_df, profile=[ConstantCpChange(delta=0.05)], methods=methods, study=study, profile_name="constant_cp_5pct" + ) + summary = leaderboard(results) + + oracle = summary[summary["method"] == "oracle"] + biased = summary[summary["method"] == "biased"] + assert np.allclose(oracle["bias"].to_numpy(), 0.0, atol=1e-6) # recovers the injected truth + assert np.allclose(biased["bias"].to_numpy(), 0.02, atol=1e-6) # off by exactly the offset + assert (summary["n_replicates"] == 2).all() + + save_path = tmp_path / "hot_campaign_curves.png" + plot_campaign_curves(summary, save_path=save_path) + assert save_path.exists() diff --git a/tests/benchmarking/harness/test_leaderboard.py b/tests/benchmarking/harness/test_leaderboard.py new file mode 100644 index 00000000..1f90b072 --- /dev/null +++ b/tests/benchmarking/harness/test_leaderboard.py @@ -0,0 +1,159 @@ +"""Tests for the leaderboard summary.""" + +from __future__ import annotations + +import math + +import pandas as pd +import pytest + +from benchmarking.harness.leaderboard import conditional_leaderboard, leaderboard + + +def _results(rows: list[dict]) -> pd.DataFrame: + base = {"method": "m", "profile": "p", "condition": "overall"} + return pd.DataFrame([{**base, **r} for r in rows]) + + +def test_one_summary_row_per_method_profile_campaign() -> None: + results = _results( + [ + {"campaign_months": 3, "signed_error": 1.0}, + {"campaign_months": 3, "signed_error": 3.0}, + {"campaign_months": 6, "signed_error": 0.5}, + ] + ) + summary = leaderboard(results) + assert len(summary) == 2 + assert set(summary["campaign_months"]) == {3, 6} + + +def test_bias_spread_score_and_n_match_metrics() -> None: + results = _results( + [ + {"campaign_months": 3, "signed_error": 1.0}, + {"campaign_months": 3, "signed_error": 3.0}, + ] + ) + row = leaderboard(results).set_index("campaign_months").loc[3] + assert row["bias"] == pytest.approx(2.0) + assert row["spread"] == pytest.approx(1.0) + assert row["score"] == pytest.approx(math.sqrt((1.0**2 + 3.0**2) / 2)) + assert row["n_replicates"] == 2 + + +def test_only_overall_condition_rows_are_summarised() -> None: + results = _results( + [ + {"campaign_months": 3, "signed_error": 2.0}, + {"campaign_months": 3, "signed_error": 100.0, "condition": "(4.0, 5.0]"}, + ] + ) + row = leaderboard(results).set_index("campaign_months").loc[3] + assert row["bias"] == pytest.approx(2.0) # the per-condition row is excluded + assert row["n_replicates"] == 1 + + +def test_mean_estimate_and_truth_are_averaged() -> None: + results = _results( + [ + {"campaign_months": 3, "signed_error": 0.01, "estimate": 0.06, "truth": 0.05}, + {"campaign_months": 3, "signed_error": -0.01, "estimate": 0.04, "truth": 0.05}, + ] + ) + row = leaderboard(results).set_index("campaign_months").loc[3] + assert row["mean_estimate"] == pytest.approx(0.05) + assert row["mean_truth"] == pytest.approx(0.05) + + +def test_wall_time_is_summed_and_averaged_per_group() -> None: + results = _results( + [ + {"campaign_months": 3, "signed_error": 0.0, "wall_time_s": 1.5}, + {"campaign_months": 3, "signed_error": 0.0, "wall_time_s": 2.5}, + {"campaign_months": 6, "signed_error": 0.0, "wall_time_s": 4.0}, + ] + ) + summary = leaderboard(results).set_index("campaign_months") + # total compute for the group, and the typical per-run cost + assert summary.loc[3, "wall_time_s_sum"] == pytest.approx(4.0) + assert summary.loc[3, "wall_time_s_mean"] == pytest.approx(2.0) + assert summary.loc[6, "wall_time_s_sum"] == pytest.approx(4.0) + assert summary.loc[6, "wall_time_s_mean"] == pytest.approx(4.0) + + +def test_methods_are_compared_side_by_side() -> None: + results = pd.DataFrame( + [ + {"method": "a", "profile": "p", "condition": "overall", "campaign_months": 6, "signed_error": 0.0}, + {"method": "b", "profile": "p", "condition": "overall", "campaign_months": 6, "signed_error": 0.1}, + ] + ) + summary = leaderboard(results) + assert set(summary["method"]) == {"a", "b"} + by_method = summary.set_index("method") + assert by_method.loc["a", "score"] == pytest.approx(0.0) + assert by_method.loc["b", "score"] == pytest.approx(0.1) + + +def test_conditional_leaderboard_groups_by_condition_bin() -> None: + df = pd.DataFrame( + { + "method": "m", + "profile": "p", + "campaign_months": 6, + "condition": ["ws", "ws", "ws", "ws"], + "condition_bin": ["(6.0, 8.0]", "(6.0, 8.0]", "(8.0, 10.0]", "(8.0, 10.0]"], + "estimate": [0.11, 0.09, 0.05, 0.05], + "truth": [0.10, 0.10, 0.05, 0.05], + "signed_error": [0.01, -0.01, 0.0, 0.0], + } + ) + lb = conditional_leaderboard(df) + assert set(lb.columns) >= { + "method", + "profile", + "campaign_months", + "condition", + "condition_bin", + "bias", + "spread", + "score", + } + row = lb[lb["condition_bin"] == "(6.0, 8.0]"].iloc[0] + assert row["bias"] == 0.0 + assert row["spread"] == pytest.approx(0.01) + + +def test_conditional_leaderboard_summarises_power_condition() -> None: + df = pd.DataFrame( + { + "method": "m", + "profile": "p", + "campaign_months": 6, + "condition": ["power", "power"], + "condition_bin": ["(230.0, 690.0]", "(230.0, 690.0]"], + "estimate": [0.06, 0.04], + "truth": [0.05, 0.05], + "signed_error": [0.01, -0.01], + } + ) + lb = conditional_leaderboard(df) + assert set(lb["condition"]) == {"power"} + assert lb.iloc[0]["spread"] == pytest.approx(0.01) + + +def test_conditional_leaderboard_ignores_overall_rows() -> None: + df = pd.DataFrame( + { + "method": "m", + "profile": "p", + "campaign_months": 6, + "condition": ["overall"], + "condition_bin": ["overall"], + "estimate": [0.1], + "truth": [0.1], + "signed_error": [0.0], + } + ) + assert conditional_leaderboard(df).empty diff --git a/tests/benchmarking/harness/test_method.py b/tests/benchmarking/harness/test_method.py new file mode 100644 index 00000000..417fcddc --- /dev/null +++ b/tests/benchmarking/harness/test_method.py @@ -0,0 +1,41 @@ +"""Tests for the thin method seam.""" + +from __future__ import annotations + +import pandas as pd + +from benchmarking.harness.method import Method, MethodInput, MethodOutput + + +def _input() -> MethodInput: + scada_df = pd.DataFrame({"x": [1.0]}, index=pd.date_range("2020-01-01", periods=1, tz="UTC")) + return MethodInput(scada_df=scada_df, test_wtg="T1", upgrade_timing=pd.Timestamp("2020-01-01", tz="UTC")) + + +def test_method_input_carries_data_test_wtg_and_timing() -> None: + mi = _input() + assert mi.test_wtg == "T1" + assert mi.upgrade_timing == pd.Timestamp("2020-01-01", tz="UTC") + assert list(mi.scada_df.columns) == ["x"] + + +def test_method_output_defaults_by_condition_to_none() -> None: + out = MethodOutput(p50_overall=0.03) + assert out.p50_overall == 0.03 + assert out.p50_by_condition is None + + +def test_a_conforming_class_satisfies_the_method_protocol() -> None: + class FixedMethod: + name = "fixed" + + def estimate(self, mi: MethodInput) -> MethodOutput: # noqa: ARG002 + return MethodOutput(p50_overall=0.0) + + method = FixedMethod() + assert isinstance(method, Method) + assert method.estimate(_input()).p50_overall == 0.0 + + +def test_a_non_conforming_object_is_not_a_method() -> None: + assert not isinstance(object(), Method) diff --git a/tests/benchmarking/harness/test_metrics.py b/tests/benchmarking/harness/test_metrics.py new file mode 100644 index 00000000..8cd9aa7a --- /dev/null +++ b/tests/benchmarking/harness/test_metrics.py @@ -0,0 +1,59 @@ +"""Tests for the accuracy/precision/score metrics.""" + +from __future__ import annotations + +import math + +import numpy as np +import pytest + +from benchmarking.harness.metrics import ErrorSummary, summarize_errors + + +def test_bias_is_the_mean_signed_error() -> None: + summary = summarize_errors([1.0, 3.0]) + assert summary.bias == pytest.approx(2.0) + + +def test_spread_is_the_population_std() -> None: + # population std of {1, 3} about mean 2 is sqrt(((1-2)^2 + (3-2)^2)/2) = 1.0 + summary = summarize_errors([1.0, 3.0]) + assert summary.spread == pytest.approx(1.0) + + +def test_score_is_rmse_and_equals_root_bias_sq_plus_spread_sq() -> None: + errors = [1.0, 3.0] + summary = summarize_errors(errors) + expected_rmse = math.sqrt((1.0**2 + 3.0**2) / 2) + assert summary.score == pytest.approx(expected_rmse) + assert summary.score == pytest.approx(math.hypot(summary.bias, summary.spread)) + + +def test_single_error_has_zero_spread_and_abs_score() -> None: + summary = summarize_errors([-0.4]) + assert summary.bias == pytest.approx(-0.4) + assert summary.spread == pytest.approx(0.0) + assert summary.score == pytest.approx(0.4) + assert summary.n == 1 + + +def test_empty_errors_give_nan_summary_with_zero_n() -> None: + summary = summarize_errors([]) + assert summary.n == 0 + assert math.isnan(summary.bias) + assert math.isnan(summary.spread) + assert math.isnan(summary.score) + + +def test_nan_errors_are_ignored() -> None: + # a replicate that produced no estimate (NaN) must not poison the summary + summary = summarize_errors([1.0, np.nan, 3.0]) + assert summary.n == 2 + assert summary.bias == pytest.approx(2.0) + + +def test_summary_is_a_frozen_dataclass() -> None: + summary = summarize_errors([1.0, 3.0]) + assert isinstance(summary, ErrorSummary) + with pytest.raises(AttributeError): + summary.bias = 0.0 # type: ignore[misc] diff --git a/tests/benchmarking/harness/test_plots.py b/tests/benchmarking/harness/test_plots.py new file mode 100644 index 00000000..01045058 --- /dev/null +++ b/tests/benchmarking/harness/test_plots.py @@ -0,0 +1,112 @@ +"""Tests for the harness campaign-length plots.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib as mpl + +mpl.use("Agg") # headless: no display needed for tests + +import matplotlib.pyplot as plt +import pandas as pd + +from benchmarking.harness.plots import plot_campaign_curves, plot_conditional_uplift + +if TYPE_CHECKING: + from pathlib import Path + + +def _summary() -> pd.DataFrame: + return pd.DataFrame( + { + "method": ["v0", "v0", "rlearner", "rlearner"], + "profile": "p", + "campaign_months": [3, 6, 3, 6], + "bias": [0.02, 0.01, 0.0, 0.0], + "spread": [0.03, 0.02, 0.02, 0.01], + "score": [0.036, 0.022, 0.02, 0.01], + "mean_truth": [0.05, 0.05, 0.05, 0.05], + "mean_estimate": [0.07, 0.06, 0.05, 0.05], + "n_replicates": 5, + } + ) + + +# fig.axes order matches the stacked panels top-to-bottom. +_UPLIFT, _BAND, _SCORE = 0, 1, 2 + + +def test_three_panels() -> None: + fig = plot_campaign_curves(_summary()) + assert len(fig.axes) == 3 + + +def test_one_line_per_method_in_bias_panel() -> None: + fig = plot_campaign_curves(_summary()) + _, labels = fig.axes[_BAND].get_legend_handles_labels() + assert set(labels) == {"v0", "rlearner"} + + +def test_top_panel_shows_true_uplift_and_each_method() -> None: + fig = plot_campaign_curves(_summary()) + _, labels = fig.axes[_UPLIFT].get_legend_handles_labels() + assert set(labels) == {"true uplift", "v0", "rlearner"} + true_line = next(line for line in fig.axes[_UPLIFT].get_lines() if line.get_label() == "true uplift") + assert true_line.get_ydata().tolist() == [5.0, 5.0] # 0.05 fraction -> 5.0 pp + + +def test_top_panel_has_spread_band() -> None: + fig = plot_campaign_curves(_summary()) + # fill_between adds a PolyCollection per method; the band makes the estimate's spread visible. + assert len(fig.axes[_UPLIFT].collections) >= 1 + + +def test_x_axis_is_campaign_length() -> None: + fig = plot_campaign_curves(_summary()) + # x-axis is shared; the label lives on the bottom panel. + assert "campaign" in fig.axes[_SCORE].get_xlabel().lower() + + +def test_x_axis_starts_at_zero() -> None: + fig = plot_campaign_curves(_summary()) + assert fig.axes[_SCORE].get_xlim()[0] == 0.0 + + +def test_y_axis_in_percentage_points() -> None: + fig = plot_campaign_curves(_summary()) + # v0 bias of 0.02/0.01 (fractions) should be plotted as 2.0/1.0 percentage points. + v0_line = next(line for line in fig.axes[_BAND].get_lines() if line.get_label() == "v0") + assert v0_line.get_ydata().tolist() == [2.0, 1.0] + + +def test_score_y_axis_floor_below_zero_so_zero_points_show() -> None: + summary = _summary() + summary.loc[summary["method"] == "rlearner", "score"] = 0.0 # an oracle-like method at 0 + fig = plot_campaign_curves(summary) + assert fig.axes[_SCORE].get_ylim()[0] < 0.0 + + +def test_saves_file(tmp_path: Path) -> None: + save_path = tmp_path / "campaign_curves.png" + plot_campaign_curves(_summary(), save_path=save_path) + assert save_path.exists() + assert save_path.stat().st_size > 0 + + +def test_plot_conditional_uplift_writes_png(tmp_path: Path) -> None: + summary = pd.DataFrame( + { + "method": ["power_model", "power_model"], + "condition": ["ws", "ws"], + "condition_bin": ["(4.0, 6.0]", "(6.0, 8.0]"], + "mean_estimate": [0.09, 0.05], + "mean_truth": [0.10, 0.05], + "bias": [-0.01, 0.0], + "spread": [0.01, 0.005], + } + ) + out = tmp_path / "cond.png" + fig = plot_conditional_uplift(summary, condition="ws", save_path=out, title="ws_dependent_cp 6mo") + assert out.exists() + plt.close(fig) diff --git a/tests/benchmarking/harness/test_public_api.py b/tests/benchmarking/harness/test_public_api.py new file mode 100644 index 00000000..89d02b66 --- /dev/null +++ b/tests/benchmarking/harness/test_public_api.py @@ -0,0 +1,24 @@ +"""The harness package exposes its main entry points at the package root.""" + +from __future__ import annotations + +from benchmarking import harness + + +def test_public_api_exports_core_entry_points() -> None: + for name in ( + "StudyConfig", + "Replicate", + "build_replicates", + "Method", + "MethodInput", + "MethodOutput", + "score_study", + "leaderboard", + "summarize_errors", + "ErrorSummary", + "CampaignWindow", + "campaign_windows", + "plot_campaign_curves", + ): + assert hasattr(harness, name), name diff --git a/tests/benchmarking/harness/test_replicates.py b/tests/benchmarking/harness/test_replicates.py new file mode 100644 index 00000000..2769e3eb --- /dev/null +++ b/tests/benchmarking/harness/test_replicates.py @@ -0,0 +1,212 @@ +"""Tests for the replicate ensemble (the precision axis).""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.harness.replicates import Replicate, StudyConfig, build_replicates, iter_replicates +from benchmarking.synthetic import HOT_COLUMNS, ConstantCpChange, ToggleSchedule +from wind_up.constants import TIMESTAMP_COL + +PROFILE = [ConstantCpChange(delta=0.05)] + + +def _base_scada(turbines: tuple[str, ...] = ("T1", "T3", "T4", "T7", "T99")) -> pd.DataFrame: + """A small multi-turbine wind farm spanning three years of daily records.""" + index = pd.date_range("2016-01-01", "2018-12-31", freq="1D", tz="UTC") + frames = [ + pd.DataFrame( + { + HOT_COLUMNS.turbine: turbine, + HOT_COLUMNS.active_power: 1000.0, + HOT_COLUMNS.wind_speed: 8.0, + HOT_COLUMNS.wind_speed_sd: 0.8, + HOT_COLUMNS.gen_rpm: 1400.0, + }, + index=index, + ) + for turbine in turbines + ] + wf_df = pd.concat(frames) + wf_df.index.name = TIMESTAMP_COL + return wf_df + + +def _study(mode: str = "prepost", n_replicates: int = 5, seed: int = 0) -> StudyConfig: + return StudyConfig( + mode=mode, + turbine_subset=["T1", "T3", "T4", "T7"], + treatment_start_range=(pd.Timestamp("2017-01-01", tz="UTC"), pd.Timestamp("2017-12-31", tz="UTC")), + min_pre_months=12, + campaign_months=[3, 6], + toggle_period=pd.Timedelta(days=14), + n_replicates=n_replicates, + seed=seed, + ) + + +def _weeks_study(mode: str = "toggle", n_replicates: int = 5, seed: int = 0) -> StudyConfig: + return StudyConfig( + mode=mode, + turbine_subset=["T1", "T3", "T4", "T7"], + treatment_start_range=(pd.Timestamp("2017-01-01", tz="UTC"), pd.Timestamp("2017-12-31", tz="UTC")), + min_pre_months=12, + campaign_weeks=[1, 2, 4, 8], + toggle_period=pd.Timedelta(days=14), + n_replicates=n_replicates, + seed=seed, + ) + + +def test_max_activity_months_is_the_longest_campaign() -> None: + assert _study().max_activity_months == 6 + + +class TestCampaignGrid: + """A study carries exactly one campaign-length grid and describes it generically.""" + + def test_months_study_exposes_months(self) -> None: + study = _study() + assert study.campaign_lengths == [3, 6] + assert study.campaign_length_col == "campaign_months" + assert study.campaign_unit == "months" + + def test_weeks_study_exposes_weeks(self) -> None: + study = _weeks_study() + assert study.campaign_lengths == [1, 2, 4, 8] + assert study.campaign_length_col == "campaign_weeks" + assert study.campaign_unit == "weeks" + + def test_neither_grid_raises(self) -> None: + with pytest.raises(ValueError, match="exactly one"): + StudyConfig( + mode="toggle", + turbine_subset=["T1"], + treatment_start_range=(pd.Timestamp("2017-01-01", tz="UTC"), pd.Timestamp("2017-12-31", tz="UTC")), + min_pre_months=12, + n_replicates=1, + ) + + def test_both_grids_raise(self) -> None: + with pytest.raises(ValueError, match="exactly one"): + StudyConfig( + mode="toggle", + turbine_subset=["T1"], + treatment_start_range=(pd.Timestamp("2017-01-01", tz="UTC"), pd.Timestamp("2017-12-31", tz="UTC")), + min_pre_months=12, + campaign_months=[3], + campaign_weeks=[2], + n_replicates=1, + ) + + def test_max_activity_months_raises_for_a_weeks_study(self) -> None: + with pytest.raises(ValueError, match="campaign_unit='weeks'"): + _ = _weeks_study().max_activity_months + + +def test_build_replicates_returns_n_replicate_records() -> None: + reps = build_replicates(_base_scada(), profile=PROFILE, study=_study(n_replicates=5)) + assert len(reps) == 5 + assert all(isinstance(r, Replicate) for r in reps) + + +def test_data_is_subset_to_turbine_subset() -> None: + reps = build_replicates(_base_scada(), profile=PROFILE, study=_study()) + present = set(reps[0].dataset.synthetic_df[HOT_COLUMNS.turbine].unique()) + assert present == {"T1", "T3", "T4", "T7"} # the other ~17 turbines dropped + + +def test_each_replicate_draws_a_test_turbine_from_the_subset() -> None: + reps = build_replicates(_base_scada(), profile=PROFILE, study=_study()) + assert all(r.test_wtg in {"T1", "T3", "T4", "T7"} for r in reps) + + +def test_treatment_start_is_a_pandas_timestamp_supporting_offset_arithmetic() -> None: + # campaign.py does `treatment_start - pd.DateOffset(...)`, which needs a pd.Timestamp + reps = build_replicates(_base_scada(), profile=PROFILE, study=_study()) + start = reps[0].treatment_start + assert isinstance(start, pd.Timestamp) + assert start.tz is not None # tz-aware, matching the SCADA index + _ = start - pd.DateOffset(months=12) # must not raise + + +def test_treatment_start_falls_within_the_configured_range() -> None: + study = _study() + reps = build_replicates(_base_scada(), profile=PROFILE, study=study) + lo, hi = study.treatment_start_range + for r in reps: + assert lo <= r.treatment_start <= hi + + +def test_draws_are_deterministic_by_seed() -> None: + base = _base_scada() + reps_a = build_replicates(base, profile=PROFILE, study=_study(seed=42)) + reps_b = build_replicates(base, profile=PROFILE, study=_study(seed=42)) + assert [(r.test_wtg, r.treatment_start) for r in reps_a] == [(r.test_wtg, r.treatment_start) for r in reps_b] + + +def test_different_seed_changes_the_draws() -> None: + base = _base_scada() + reps_a = build_replicates(base, profile=PROFILE, study=_study(seed=1)) + reps_b = build_replicates(base, profile=PROFILE, study=_study(seed=2)) + assert [(r.test_wtg, r.treatment_start) for r in reps_a] != [(r.test_wtg, r.treatment_start) for r in reps_b] + + +def test_prepost_replicate_upgrade_timing_is_the_treatment_start() -> None: + reps = build_replicates(_base_scada(), profile=PROFILE, study=_study(mode="prepost")) + r = reps[0] + assert r.upgrade_timing == r.treatment_start + + +def test_toggle_replicate_builds_schedule_with_start_and_period() -> None: + study = _study(mode="toggle") + reps = build_replicates(_base_scada(), profile=PROFILE, study=study) + timing = reps[0].upgrade_timing + assert isinstance(timing, ToggleSchedule) + assert timing.start == reps[0].treatment_start + assert timing.period == study.toggle_period + + +def test_replicate_true_uplift_delegates_to_its_dataset() -> None: + reps = build_replicates(_base_scada(), profile=PROFILE, study=_study()) + r = reps[0] + via_replicate = r.true_uplift() + via_dataset = r.dataset.true_uplift(test_wtg=r.test_wtg) + assert via_replicate.overall == via_dataset.overall + + +def test_injected_upgrade_actually_changed_the_test_turbine() -> None: + reps = build_replicates(_base_scada(), profile=PROFILE, study=_study()) + r = reps[0] + syn = r.dataset.synthetic_df + orig = r.dataset.original_df + test_syn = syn[syn[HOT_COLUMNS.turbine] == r.test_wtg][HOT_COLUMNS.active_power].to_numpy() + test_orig = orig[orig[HOT_COLUMNS.turbine] == r.test_wtg][HOT_COLUMNS.active_power].to_numpy() + assert np.any(test_syn != test_orig) # the profile left a mark + + +# --- streaming --------------------------------------------------------------------------------- + + +def test_iter_replicates_yields_the_same_replicates_as_build_replicates() -> None: + base = _base_scada() + study = _study(n_replicates=4) + built = build_replicates(base, profile=PROFILE, study=study) + streamed = list(iter_replicates(base, profile=PROFILE, study=study)) + + assert len(streamed) == len(built) + for a, b in zip(built, streamed, strict=True): + assert a.test_wtg == b.test_wtg + assert a.treatment_start == b.treatment_start + assert a.replicate_id == b.replicate_id + pd.testing.assert_frame_equal(a.synthetic_df, b.synthetic_df) + + +def test_iter_replicates_is_lazy() -> None: + """The point of streaming: nothing is generated until asked for, so memory stays bounded.""" + base = _base_scada() + stream = iter_replicates(base, profile=PROFILE, study=_study(n_replicates=100)) + first = next(stream) # would be unusably slow and huge if all 100 were built up front + assert first.replicate_id == 0 diff --git a/tests/benchmarking/harness/test_scoring.py b/tests/benchmarking/harness/test_scoring.py new file mode 100644 index 00000000..8d21e543 --- /dev/null +++ b/tests/benchmarking/harness/test_scoring.py @@ -0,0 +1,303 @@ +"""Tests for the scoring orchestrator and the two-part fairness guarantee.""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.harness.campaign import campaign_windows +from benchmarking.harness.replicates import StudyConfig, build_replicates +from benchmarking.harness.scoring import _merge_diagnostics, score_one, score_study, truth_mask +from benchmarking.synthetic import HOT_COLUMNS, ConstantCpChange +from wind_up.constants import TIMESTAMP_COL + +from .stubs import BiasedMethod, ConditionalOracleMethod, OracleMethod, RecordingMethod, UncertainMethod + +PROFILE = [ConstantCpChange(delta=0.05)] + + +def _base_scada(turbines: tuple[str, ...] = ("T1", "T3", "T4", "T7")) -> pd.DataFrame: + index = pd.date_range("2016-01-01", "2018-12-31", freq="1D", tz="UTC") + frames = [ + pd.DataFrame( + { + HOT_COLUMNS.turbine: turbine, + HOT_COLUMNS.active_power: 1000.0, + HOT_COLUMNS.wind_speed: 8.0, + HOT_COLUMNS.wind_speed_sd: 0.8, + HOT_COLUMNS.gen_rpm: 1400.0, + }, + index=index, + ) + for turbine in turbines + ] + wf_df = pd.concat(frames) + wf_df.index.name = TIMESTAMP_COL + return wf_df + + +def _study(mode: str = "prepost", n_replicates: int = 4, seed: int = 0) -> StudyConfig: + return StudyConfig( + mode=mode, + turbine_subset=["T1", "T3", "T4", "T7"], + treatment_start_range=(pd.Timestamp("2017-01-01", tz="UTC"), pd.Timestamp("2017-12-31", tz="UTC")), + min_pre_months=12, + campaign_months=[3, 6], + toggle_period=pd.Timedelta(days=14), + n_replicates=n_replicates, + seed=seed, + ) + + +def test_results_have_one_row_per_method_replicate_and_campaign_length() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study(n_replicates=4)) + # 4 replicates x 2 campaign lengths x 1 method + assert len(results) == 4 * 2 + assert set(results["campaign_months"]) == {3, 6} + expected_columns = {"method", "profile", "replicate", "campaign_months", "estimate", "truth", "signed_error"} + assert set(results.columns) >= expected_columns + + +def test_results_record_wall_time_per_run() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study(n_replicates=1)) + assert "wall_time_s" in results.columns + assert results["wall_time_s"].notna().all() + assert (results["wall_time_s"] >= 0).all() + + +def test_results_record_the_window_boundaries_per_replicate() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study(n_replicates=1)) + for col in ("treatment_start", "baseline_start", "activity_end"): + assert col in results.columns + assert results[col].notna().all() + # baseline_start = treatment_start - min_pre_months (12), activity_end = treatment_start + campaign_months + row3 = results[results["campaign_months"] == 3].iloc[0] + assert row3["baseline_start"] == row3["treatment_start"] - pd.DateOffset(months=12) + assert row3["activity_end"] == row3["treatment_start"] + pd.DateOffset(months=3) + + +def test_oracle_method_has_near_zero_signed_error() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study()) + assert np.allclose(results["signed_error"].to_numpy(), 0.0, atol=1e-9) + + +def test_biased_method_signed_error_equals_offset() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[BiasedMethod(base, offset=0.02)], study=_study()) + assert np.allclose(results["signed_error"].to_numpy(), 0.02, atol=1e-9) + + +def test_truth_is_recomputed_per_campaign_length_not_shared() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study()) + # constant power -> uplift is ~equal across lengths, but truth must be present per row + assert results["truth"].notna().all() + assert (results["truth"] > 0).all() # +5% Cp raises region-2 power + + +def test_profile_name_is_recorded() -> None: + base = _base_scada() + results = score_study( + base, profile=PROFILE, methods=[OracleMethod(base)], study=_study(), profile_name="constant_cp_5pct" + ) + assert set(results["profile"]) == {"constant_cp_5pct"} + + +def test_fairness_every_method_sees_identical_method_inputs() -> None: + base = _base_scada() + rec_a = RecordingMethod(name="a") + rec_b = RecordingMethod(name="b") + score_study(base, profile=PROFILE, methods=[rec_a, rec_b], study=_study()) + + assert len(rec_a.seen) == len(rec_b.seen) > 0 + for mi_a, mi_b in zip(rec_a.seen, rec_b.seen, strict=True): + assert mi_a.test_wtg == mi_b.test_wtg + assert mi_a.upgrade_timing == mi_b.upgrade_timing + pd.testing.assert_frame_equal(mi_a.scada_df, mi_b.scada_df) + + +def test_fairness_shorter_campaign_input_is_a_prefix_of_the_longer() -> None: + base = _base_scada() + rec = RecordingMethod() + score_study(base, profile=PROFILE, methods=[rec], study=_study(n_replicates=1)) + # one replicate, lengths [3, 6] in order + short_mi, long_mi = rec.seen[0], rec.seen[1] + assert short_mi.scada_df.index.min() == long_mi.scada_df.index.min() # same baseline start + assert short_mi.scada_df.index.max() <= long_mi.scada_df.index.max() # shorter activity + assert len(short_mi.scada_df) < len(long_mi.scada_df) + + +def test_toggle_study_scores_without_error() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study(mode="toggle")) + assert len(results) == 4 * 2 + assert np.allclose(results["signed_error"].to_numpy(), 0.0, atol=1e-9) + + +def test_two_methods_scored_side_by_side() -> None: + base = _base_scada() + results = score_study( + base, profile=PROFILE, methods=[OracleMethod(base), BiasedMethod(base, offset=0.01)], study=_study() + ) + assert set(results["method"]) == {"oracle", "biased"} + oracle_err = results[results["method"] == "oracle"]["signed_error"].to_numpy() + biased_err = results[results["method"] == "biased"]["signed_error"].to_numpy() + assert np.allclose(oracle_err, 0.0, atol=1e-9) + assert np.allclose(biased_err, 0.01, atol=1e-9) + + +def test_on_method_complete_fires_per_method_with_only_that_methods_rows() -> None: + base = _base_scada() + seen: list[tuple[str, pd.DataFrame]] = [] + methods = [OracleMethod(base), BiasedMethod(base, offset=0.01)] + results = score_study( + base, + profile=PROFILE, + methods=methods, + study=_study(), + on_method_complete=lambda name, df: seen.append((name, df)), + ) + # one call per method, in method order, each carrying only that method's rows + assert [name for name, _ in seen] == ["oracle", "biased"] + for name, df in seen: + assert set(df["method"]) == {name} + assert len(df) == 4 * 2 # 4 replicates x 2 campaign lengths + # the callback never changes the returned frame: the slices concatenate back to it exactly + rebuilt = pd.concat([df for _, df in seen], ignore_index=True) + pd.testing.assert_frame_equal(rebuilt, results) + + +def test_on_method_complete_is_optional() -> None: + base = _base_scada() + with_cb = score_study( + base, profile=PROFILE, methods=[OracleMethod(base)], study=_study(), on_method_complete=lambda _n, _d: None + ) + without_cb = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study()) + # wall_time_s is measured per run, so it differs between two runs; the callback must not change + # anything else. + drop = ["wall_time_s"] + pd.testing.assert_frame_equal(with_cb.drop(columns=drop), without_cb.drop(columns=drop)) + + +def test_conditional_rows_are_emitted_with_truth_and_near_zero_error() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[ConditionalOracleMethod(base)], study=_study(n_replicates=1)) + assert "condition_bin" in results.columns + overall = results[results["condition"] == "overall"] + assert (overall["condition_bin"] == "overall").all() + cond = results[results["condition"].isin(["ws", "ti", "power"])] + assert set(cond["condition"]) == {"ws", "ti", "power"} + # populated bins: a per-bin oracle must match per-bin truth (power too, via the rating-scaled edges) + populated = cond[cond["truth"].notna() & cond["estimate"].notna()] + assert len(populated) > 0 + assert (populated["condition"] == "power").any() + assert np.allclose(populated["signed_error"].to_numpy(), 0.0, atol=1e-9) + + +def test_overall_only_method_emits_no_conditional_rows() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study(n_replicates=1)) + assert set(results["condition"]) == {"overall"} + + +# --- uncertainty passthrough ------------------------------------------------------------------- + + +def test_sigma_is_nan_for_a_method_that_reports_no_uncertainty() -> None: + base = _base_scada() + results = score_study(base, profile=PROFILE, methods=[OracleMethod(base)], study=_study(n_replicates=1)) + assert "sigma" in results.columns + assert results["sigma"].isna().all() + + +def test_sigma_reaches_the_results_for_overall_and_per_bin_rows() -> None: + base = _base_scada() + results = score_study( + base, profile=PROFILE, methods=[UncertainMethod(base, sigma=0.03)], study=_study(n_replicates=1) + ) + overall = results[results["condition"] == "overall"] + per_bin = results[results["condition"] != "overall"] + assert (overall["sigma"] == 0.03).all() + # the stub reports twice the overall sigma per bin, so the two channels are told apart + assert (per_bin["sigma"] == 0.06).all() + + +def test_uncertainty_diagnostics_are_carried_to_every_row() -> None: + base = _base_scada() + results = score_study( + base, + profile=PROFILE, + methods=[UncertainMethod(base, diagnostic_columns={"n_blocks": 7.0, "frac_resamples_finite": 0.5})], + study=_study(n_replicates=1), + ) + assert (results["n_blocks"] == 7.0).all() + assert (results["frac_resamples_finite"] == 0.5).all() + + +def test_a_diagnostics_column_clashing_with_a_harness_column_raises() -> None: + """Silently overwriting `estimate` or `truth` would corrupt the very thing the row reports.""" + base = _base_scada() + with pytest.raises(ValueError, match="clash with columns the harness owns"): + score_study( + base, + profile=PROFILE, + methods=[UncertainMethod(base, diagnostic_columns={"estimate": 0.0})], + study=_study(n_replicates=1), + ) + + +def test_diagnostics_missing_a_key_column_raises_a_clear_error() -> None: + """A method returning a frame not keyed by (condition, condition_bin) gets a contract error, + not an opaque KeyError from the row lookup. + """ + diagnostics = pd.DataFrame([{"condition": "overall", "n_blocks": 7}]) # no condition_bin + with pytest.raises(ValueError, match="must be keyed by"): + _merge_diagnostics([], diagnostics) + + +# --- score_one is the unit score_study is built from ------------------------------------------- + + +def test_score_one_reproduces_score_study_row_for_row() -> None: + base = _base_scada() + study = _study(n_replicates=2) + method = UncertainMethod(base) + from_study = score_study(base, profile=PROFILE, methods=[method], study=study, profile_name="p") + + replicates = build_replicates(base, profile=PROFILE, study=study) + rows: list[dict] = [] + for replicate in replicates: + for window in campaign_windows( + replicate.treatment_start, + min_pre_months=study.min_pre_months, + campaign_months=study.campaign_months, + campaign_weeks=study.campaign_weeks, + data_start=base.index.min(), + data_end=base.index.max(), + ): + mask = truth_mask(replicate, window) + truth = replicate.true_uplift(mask=mask).overall + rows.extend(score_one(method, replicate=replicate, window=window, truth=truth, mask=mask, profile_name="p")) + from_one = pd.DataFrame(rows) + + drop = ["wall_time_s"] # measured per call, so it differs between two runs + pd.testing.assert_frame_equal( + from_study.drop(columns=drop).reset_index(drop=True), from_one.drop(columns=drop).reset_index(drop=True) + ) + + +def test_duplicate_diagnostics_keys_raise_rather_than_silently_win() -> None: + """Keeping the last row would drop a diagnostic without a word; the frame is keyed, so say so.""" + diagnostics = pd.DataFrame( + [ + {"condition": "power", "condition_bin": "(0, 100]", "n_blocks": 7}, + {"condition": "power", "condition_bin": "(0, 100]", "n_blocks": 9}, + ] + ) + with pytest.raises(ValueError, match="duplicate"): + _merge_diagnostics([], diagnostics) diff --git a/tests/benchmarking/harness/test_toggle.py b/tests/benchmarking/harness/test_toggle.py new file mode 100644 index 00000000..de984820 --- /dev/null +++ b/tests/benchmarking/harness/test_toggle.py @@ -0,0 +1,180 @@ +"""Tests for the shared toggle interpreter consumed by all benchmarking methods. + +The interpreter turns a toggle campaign (a periodic ``ToggleSchedule`` or an explicit, possibly +irregular ``toggle_df``) into the three row-sets methods need: ``upgraded`` (the "on" segment), +``campaign_baseline`` (the strict "off" segment, for the on/off comparison and matching) and +``training_baseline`` (the lenient pre-first-on U off segment, for counterfactual model fitting). +Rows that are neither on nor a training-baseline row are excluded from every segment. +""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.harness.toggle import ( + ToggleRowSets, + build_toggle_df, + is_toggle, + resolve_toggle, + toggle_upgrade_start, +) +from benchmarking.synthetic import ToggleSchedule, treated_mask + + +def _toggle_df(index: pd.DatetimeIndex, on: list[bool], off: list[bool]) -> pd.DataFrame: + return pd.DataFrame({"toggle_on": on, "toggle_off": off}, index=index) + + +def test_is_toggle_recognises_schedules_and_toggle_frames_but_not_timestamps() -> None: + index = pd.date_range("2025-01-01", periods=3, freq="10min") + assert is_toggle(ToggleSchedule(period=pd.Timedelta(hours=1))) is True + assert is_toggle(_toggle_df(index, [True, False, False], [False, True, False])) is True + assert is_toggle(pd.Timestamp("2025-01-01")) is False + + +def test_resolve_toggle_explicit_frame_excludes_neither_rows_from_every_segment() -> None: + """A 'neither' row (both flags False) is in neither upgraded, campaign nor training baseline.""" + index = pd.date_range("2025-01-01", periods=4, freq="10min") + # on off neither on + toggle_df = _toggle_df(index, on=[True, False, False, True], off=[False, True, False, False]) + + rows = resolve_toggle(toggle_df, index) + + assert list(rows.upgraded) == [True, False, False, True] + assert list(rows.campaign_baseline) == [False, True, False, False] + # training_baseline = (index < first_upgraded) U off; first upgraded is index[0], so no pre rows. + assert list(rows.training_baseline) == [False, True, False, False] + # the neither row (position 2) is excluded everywhere. + assert not rows.upgraded[2] + assert not rows.campaign_baseline[2] + assert not rows.training_baseline[2] + + +def test_resolve_toggle_training_baseline_includes_pre_first_upgraded_rows() -> None: + """Rows before the first 'on' row join the lenient training baseline but not the campaign one.""" + index = pd.date_range("2025-01-01", periods=4, freq="10min") + # pre-campaign 'neither', then off, then on, then off + toggle_df = _toggle_df(index, on=[False, False, True, False], off=[False, True, False, True]) + + rows = resolve_toggle(toggle_df, index) + + assert list(rows.upgraded) == [False, False, True, False] + assert list(rows.campaign_baseline) == [False, True, False, True] + # position 0 is before the first upgraded row (index[2]) -> in training baseline only. + assert list(rows.training_baseline) == [True, True, False, True] + + +def test_resolve_toggle_rejects_simultaneous_on_and_off() -> None: + index = pd.date_range("2025-01-01", periods=2, freq="10min") + bad = _toggle_df(index, on=[True, False], off=[True, False]) + with pytest.raises(ValueError, match="toggle_on and toggle_off"): + resolve_toggle(bad, index) + + +def test_resolve_toggle_rejects_toggle_df_missing_a_flag_column() -> None: + index = pd.date_range("2025-01-01", periods=2, freq="10min") + missing_off = pd.DataFrame({"toggle_on": [True, False]}, index=index) + with pytest.raises(ValueError, match=r"missing.*toggle_off"): + resolve_toggle(missing_off, index) + + +def test_resolve_toggle_rejects_toggle_df_with_duplicate_timestamps() -> None: + """A downstream frame with duplicate timestamps fails with a clear error, not pandas' opaque one.""" + ts = pd.Timestamp("2025-01-01") + dup = _toggle_df(pd.DatetimeIndex([ts, ts]), on=[True, False], off=[False, True]) + with pytest.raises(ValueError, match="index must be unique"): + resolve_toggle(dup, pd.date_range("2025-01-01", periods=2, freq="10min")) + + +def test_resolve_toggle_reindexes_onto_a_repeated_long_frame_index() -> None: + """A long frame repeats each timestamp per turbine; the labels broadcast to every row.""" + ts = pd.date_range("2025-01-01", periods=3, freq="10min") + toggle_df = _toggle_df(ts, on=[True, False, True], off=[False, True, False]) + long_index = ts.repeat(2) # two turbines per timestamp + + rows = resolve_toggle(toggle_df, long_index) + + assert list(rows.upgraded) == [True, True, False, False, True, True] + assert list(rows.campaign_baseline) == [False, False, True, True, False, False] + + +def test_resolve_toggle_prepost_timestamp_splits_on_the_changeover() -> None: + index = pd.date_range("2025-01-01", periods=4, freq="10min") + changeover = index[2] + + rows = resolve_toggle(changeover, index) + + assert list(rows.upgraded) == [False, False, True, True] + # prepost: both baselines are the pre-changeover rows. + assert list(rows.campaign_baseline) == [True, True, False, False] + assert list(rows.training_baseline) == [True, True, False, False] + + +def test_resolve_toggle_schedule_matches_legacy_treated_mask_definitions() -> None: + """On a periodic ToggleSchedule the row-sets equal the legacy binary masks (behaviour-preserving). + + upgraded == treated_mask; campaign_baseline == ~treated & (index>=start); + training_baseline == ~treated (pre-campaign U off). + """ + index = pd.date_range("2025-01-01", periods=48, freq="30min") + start = index[12] + schedule = ToggleSchedule(period=pd.Timedelta(hours=2), start=start) + + rows = resolve_toggle(schedule, index) + legacy_treated = np.asarray(treated_mask(index, schedule)) + at_or_after_start = np.asarray(index >= start) + + assert np.array_equal(rows.upgraded, legacy_treated) + assert np.array_equal(rows.campaign_baseline, ~legacy_treated & at_or_after_start) + assert np.array_equal(rows.training_baseline, ~legacy_treated) + + +def test_build_toggle_df_from_schedule_is_neither_before_start_then_exactly_one_flag() -> None: + index = pd.date_range("2025-01-01", periods=24, freq="1h") + start = index[6] + schedule = ToggleSchedule(period=pd.Timedelta(hours=4), start=start) + + toggle_df = build_toggle_df(index, schedule) + + assert list(toggle_df.columns) == ["toggle_on", "toggle_off"] + pre = index < start + # before start: neither flag set. + assert not toggle_df.loc[pre, "toggle_on"].any() + assert not toggle_df.loc[pre, "toggle_off"].any() + # from start: exactly one flag true per row. + post = toggle_df.loc[~pre] + assert (post["toggle_on"] ^ post["toggle_off"]).all() + + +def test_resolve_toggle_returns_row_sets_dataclass() -> None: + index = pd.date_range("2025-01-01", periods=2, freq="10min") + rows = resolve_toggle(_toggle_df(index, [True, False], [False, True]), index) + assert isinstance(rows, ToggleRowSets) + + +def test_toggle_upgrade_start_prepost_is_the_changeover() -> None: + index = pd.date_range("2025-01-01", periods=4, freq="10min") + assert toggle_upgrade_start(index[2], index) == index[2] + + +def test_toggle_upgrade_start_schedule_is_its_start() -> None: + index = pd.date_range("2025-01-01", periods=48, freq="30min") + schedule = ToggleSchedule(period=pd.Timedelta(hours=2), start=index[12]) + assert toggle_upgrade_start(schedule, index) == index[12] + + +def test_toggle_upgrade_start_frame_is_first_upgraded_row_within_index() -> None: + """The start aligns to the analysis index: on-rows the analysis never sees do not count.""" + index = pd.date_range("2025-01-01", periods=4, freq="10min") + # An earlier on-row before the analysis window would win a raw ``.min()``, but is not in ``index``. + frame_index = pd.DatetimeIndex([pd.Timestamp("2024-12-31"), *index]) + toggle_df = _toggle_df(frame_index, on=[True, False, True, False, False], off=[False, True, False, True, True]) + assert toggle_upgrade_start(toggle_df, index) == index[1] + + +def test_toggle_upgrade_start_frame_never_on_falls_back_to_index_min() -> None: + index = pd.date_range("2025-01-01", periods=3, freq="10min") + toggle_df = _toggle_df(index, on=[False, False, False], off=[True, True, True]) + assert toggle_upgrade_start(toggle_df, index) == index.min() diff --git a/tests/benchmarking/synthetic/__init__.py b/tests/benchmarking/synthetic/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/benchmarking/synthetic/sources/__init__.py b/tests/benchmarking/synthetic/sources/__init__.py new file mode 100644 index 00000000..e69de29b diff --git a/tests/benchmarking/synthetic/sources/test_hill_of_towie.py b/tests/benchmarking/synthetic/sources/test_hill_of_towie.py new file mode 100644 index 00000000..82126b02 --- /dev/null +++ b/tests/benchmarking/synthetic/sources/test_hill_of_towie.py @@ -0,0 +1,180 @@ +"""Offline tests for the Hill of Towie source adapter. + +These cover the pure SCADA transforms: the source-native wide-to-long reshape +(:func:`scada_wide_to_long`) the rest of the pipeline sees, and the v0-only wind-up-format +on-ramp (:func:`long_to_wind_up_format`) that aliases the columns and derives ``PitchAngleMean`` +/ ``ShutdownDuration``. The network path (Zenodo download via ``load_hot_scada``) is exercised +separately by a ``slow``-marked test that is excluded from the default offline suite. +""" + +from __future__ import annotations + +import json +from typing import TYPE_CHECKING +from unittest.mock import MagicMock, patch + +import numpy as np +import pandas as pd + +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.sources.hill_of_towie import ( + download_zenodo_data, + long_to_wind_up_format, + scada_wide_to_long, +) +from wind_up.constants import TIMESTAMP_COL, DataColumns + +if TYPE_CHECKING: + from pathlib import Path + +TIMEBASE_S = 600 + +# Source-native 10-min tag names, as the loader emits them before any v0 aliasing. +_F_ACTIVE_POWER = "wtc_ActPower_mean" +_F_ACTIVE_POWER_SD = "wtc_ActPower_stddev" +_F_WIND_SPEED = "wtc_AcWindSp_mean" +_F_WIND_SPEED_SD = "wtc_AcWindSp_stddev" +_F_YAW_MEAN = "wtc_NacelPos_mean" +_F_YAW_MIN = "wtc_NacelPos_min" +_F_YAW_MAX = "wtc_NacelPos_max" +_F_GEN_RPM = "wtc_GenRpm_mean" +_F_PITCH_A = "wtc_PitcPosA_mean" +_F_PITCH_B = "wtc_PitcPosB_mean" +_F_PITCH_C = "wtc_PitcPosC_mean" +_F_TIME_READY = "wtc_ScReToOp_timeon" + + +def _wide_scada_df(*, turbines: tuple[str, ...] = ("T01", "T02"), periods: int = 6) -> pd.DataFrame: + """Build a fabricated wide two-level SCADA frame as ``load_hot_10min_data`` emits it. + + Level 0 of the columns is the turbine name (index name ``StationId``); level 1 is the + source-native ``wtc_*`` tag name. The row index is the start-format timestamp. + """ + index = pd.date_range("2020-01-01", periods=periods, freq="10min", tz="UTC") + index.name = TIMESTAMP_COL + + fields = { + _F_ACTIVE_POWER: np.linspace(100.0, 600.0, periods), + _F_ACTIVE_POWER_SD: np.full(periods, 10.0), + _F_WIND_SPEED: np.linspace(5.0, 10.0, periods), + _F_WIND_SPEED_SD: np.full(periods, 1.0), + _F_YAW_MEAN: np.full(periods, 180.0), + _F_YAW_MIN: np.full(periods, 175.0), + _F_YAW_MAX: np.full(periods, 185.0), + _F_GEN_RPM: np.linspace(1000.0, 1500.0, periods), + _F_PITCH_A: np.full(periods, 1.0), + _F_PITCH_B: np.full(periods, 2.0), + _F_PITCH_C: np.full(periods, 3.0), + _F_TIME_READY: np.full(periods, float(TIMEBASE_S)), + } + # Both turbines share identical values at each timestamp but vary over time. Stuck + # detection must compare each turbine to its OWN previous record (grouped by turbine), + # so two turbines that merely match each other are not mistaken for frozen data. + per_turbine = {turbine: pd.DataFrame(fields, index=index) for turbine in turbines} + wide = pd.concat(per_turbine, axis=1) + wide.columns = wide.columns.set_names(["StationId", None]) + return wide + + +def test_scada_wide_to_long_emits_source_native_long_format() -> None: + """The wide two-level frame becomes a single-level long frame keyed by the source schema.""" + long_df = scada_wide_to_long(_wide_scada_df()) + + assert long_df.columns.nlevels == 1 + assert long_df.index.name == TIMESTAMP_COL + assert set(long_df[HOT_COLUMNS.turbine].unique()) == {"T01", "T02"} + for col in (HOT_COLUMNS.active_power, HOT_COLUMNS.wind_speed, HOT_COLUMNS.wind_speed_sd, HOT_COLUMNS.gen_rpm): + assert col in long_df.columns + # The v0-only derived columns are not added by the source-native reshape. + assert DataColumns.pitch_angle_mean not in long_df.columns + assert DataColumns.shutdown_duration not in long_df.columns + + +def test_long_to_wind_up_format_aliases_and_computes_mean_pitch() -> None: + """The v0 on-ramp aliases the source columns and derives PitchAngleMean.""" + wind_up_df = long_to_wind_up_format(scada_wide_to_long(_wide_scada_df())) + + assert DataColumns.active_power_mean in wind_up_df.columns + assert np.allclose(wind_up_df[DataColumns.pitch_angle_mean], 2.0) + + +def test_calc_shutdown_duration_zero_when_fully_available() -> None: + """Two turbines that share values per timestamp but vary over time are all available. + + Stuck detection compares each turbine to its own previous record, so turbines that + merely match each other at the same timestamp are not flagged as frozen. + """ + wind_up_df = long_to_wind_up_format(scada_wide_to_long(_wide_scada_df())) + assert DataColumns.shutdown_duration in wind_up_df.columns + assert np.allclose(wind_up_df[DataColumns.shutdown_duration], 0.0) + + +def test_calc_shutdown_duration_flags_stuck_data() -> None: + """Repeated (stuck) rows above the low-wind threshold are flagged as full downtime.""" + wide = _wide_scada_df(turbines=("T01",), periods=4) + # Freeze T01 to identical values across all rows (stuck), at a wind speed above 1.5 m/s. + for field in wide.columns.get_level_values(1).unique(): + wide.loc[:, ("T01", field)] = wide.iloc[0].loc[("T01", field)] + wind_up_df = long_to_wind_up_format(scada_wide_to_long(wide)) + # First row has no prior to diff against; subsequent stuck rows are full downtime. + assert np.allclose(wind_up_df[DataColumns.shutdown_duration].iloc[1:], TIMEBASE_S) + + +def test_calc_shutdown_duration_flags_only_the_frozen_turbine() -> None: + """In a multi-turbine frame, only the turbine whose own data is frozen is flagged.""" + wide = _wide_scada_df(turbines=("T01", "T02"), periods=4) + # Freeze T01 across all rows (stuck); T02 keeps varying over time. + for field in wide.columns.get_level_values(1).unique(): + wide.loc[:, ("T01", field)] = wide.iloc[0].loc[("T01", field)] + wind_up_df = long_to_wind_up_format(scada_wide_to_long(wide)) + t01 = wind_up_df.loc[wind_up_df[DataColumns.turbine_name] == "T01", DataColumns.shutdown_duration].to_numpy() + t02 = wind_up_df.loc[wind_up_df[DataColumns.turbine_name] == "T02", DataColumns.shutdown_duration].to_numpy() + assert np.allclose(t01[1:], TIMEBASE_S) # frozen turbine -> downtime after the first row + assert np.allclose(t02, 0.0) # varying turbine -> available throughout + + +def test_download_routes_through_one_session_closed_once(tmp_path: Path) -> None: + """Every Zenodo request goes through a single Session that is closed exactly once. + + Regression guard for leaked SSL sockets: a per-call ``requests.get`` closes its + transient connection pool before the streamed response's socket is released back to + it, so the socket lingers until GC and trips ``filterwarnings = error`` via + ResourceWarning. A single Session whose pool is closed on exit releases every socket + deterministically. Asserting one Session, closed once, encodes exactly that fix. + """ + # Pre-cache the metadata so this stays offline; the file-download branch (the leak + # source) still has to route its streamed get through the shared Session. + file_entry = {"key": "small.txt", "size": 10, "links": {"self": "https://zenodo.test/small.txt"}} + (tmp_path / "zenodo_dataset_metadata.json").write_text(json.dumps({"files": [file_entry]})) + + response = MagicMock() + response.status_code = 200 + response.iter_content.return_value = [b"0123456789"] + response.__enter__.return_value = response + response.__exit__.return_value = False + + session = MagicMock() + session.get.return_value = response + session.__enter__.return_value = session + session.__exit__.return_value = False + + def _banned_get(*_args: object, **_kwargs: object) -> object: + msg = "per-call requests.get leaks sockets; route every request through the shared Session" + raise AssertionError(msg) + + with ( + patch("requests.Session", return_value=session) as session_cls, + patch("requests.get", side_effect=_banned_get), + ): + download_zenodo_data(record_id="123", output_dir=tmp_path) + + session_cls.assert_called_once() # exactly one Session for the whole download + session.__enter__.assert_called_once() + session.__exit__.assert_called_once() # pool (and its sockets) closed deterministically + session.get.assert_called_once_with( + file_entry["links"]["self"], + stream=True, + timeout=(10, 60), + headers={}, + ) + assert (tmp_path / "small.txt").read_bytes() == b"0123456789" diff --git a/tests/benchmarking/synthetic/sources/test_hill_of_towie_cache.py b/tests/benchmarking/synthetic/sources/test_hill_of_towie_cache.py new file mode 100644 index 00000000..67efba01 --- /dev/null +++ b/tests/benchmarking/synthetic/sources/test_hill_of_towie_cache.py @@ -0,0 +1,109 @@ +"""Tests for the per-(year, turbine) parquet cache in the Hill of Towie loader. + +The Hill of Towie Zenodo record is fixed as one zip per year, so the loader caches each +*turbine-year* once and reuses it for any window or turbine subset. These tests fabricate +minimal year zips (so they run offline, no Zenodo) and check that: + +- one parquet is written per (year, requested turbine), and only for the requested turbines; +- a repeat call reads the cache and does not re-unpack; +- growing the turbine subset only unpacks the newly requested turbines. +""" + +from __future__ import annotations + +from typing import TYPE_CHECKING +from unittest.mock import patch +from zipfile import ZipFile + +import pandas as pd + +from benchmarking.synthetic.sources import hill_of_towie as hot +from wind_up.constants import DataColumns + +if TYPE_CHECKING: + from pathlib import Path + +_APM_FIELD = hot.WPSBackupFileField( + alias=DataColumns.active_power_mean, field_name="wtc_ActPower_mean", table_name="tblSCTurGrid" +) +_WINDOW = { + "start_dt": pd.Timestamp("2020-01-01", tz="UTC"), + "end_dt_excl": pd.Timestamp("2020-03-01", tz="UTC"), +} + + +def _month_csv(*, year: int, month: int, serials: list[int]) -> str: + """An end-format SCADA CSV for one month of one table, a few timestamps per turbine.""" + end_times = pd.date_range(f"{year}-{month:02d}-01 00:10", periods=4, freq="10min") + rows = [ + {"TimeStamp": ts, "StationId": s, "wtc_ActPower_mean": 100.0 + i + s} + for i, ts in enumerate(end_times) + for s in serials + ] + return pd.DataFrame(rows).to_csv(index=False) + + +def _write_year_zip(*, data_dir: Path, year: int, months: list[int], serials: list[int]) -> None: + """Write a fabricated ``{year}.zip`` holding one ``tblSCTurGrid`` CSV per month.""" + with ZipFile(data_dir / f"{year}.zip", "w") as zf: + for m in months: + zf.writestr(f"tblSCTurGrid_{year}_{m:02d}.csv", _month_csv(year=year, month=m, serials=serials)) + + +def _load(*, data_dir: Path, cache_dir: Path, wtg_numbers: list[int]) -> pd.DataFrame: + return hot.load_hot_10min_data( + data_dir=data_dir, + wtg_numbers=wtg_numbers, + custom_fields=[_APM_FIELD], + cache_dir=cache_dir, + **_WINDOW, + ) + + +def _dirs(tmp_path: Path) -> tuple[Path, Path]: + data_dir = tmp_path / "data" + cache_dir = tmp_path / "cache" + data_dir.mkdir() + serials = [hot._HOT_SERIAL_OFFSET + n for n in (1, 3)] # noqa: SLF001 + _write_year_zip(data_dir=data_dir, year=2020, months=[1, 2], serials=serials) + return data_dir, cache_dir + + +def test_writes_one_cache_file_per_requested_turbine(tmp_path: Path) -> None: + """Loading only T01 caches T01's year and leaves the other turbine's year un-cached.""" + data_dir, cache_dir = _dirs(tmp_path) + + wide = _load(data_dir=data_dir, cache_dir=cache_dir, wtg_numbers=[1]) + + assert list(cache_dir.glob("hot10min_2020_T01_*.parquet")) + assert not list(cache_dir.glob("hot10min_2020_T03_*.parquet")) + assert {col[0] for col in wide.columns} == {"T01"} + # Columns keep their source-native tag names (no v0 aliasing at load time). + assert (("T01", _APM_FIELD.field_name)) in wide.columns + + +def test_repeat_call_reuses_cache_without_unpacking(tmp_path: Path) -> None: + """A second identical call serves the parquet cache and never re-reads the zip.""" + data_dir, cache_dir = _dirs(tmp_path) + _load(data_dir=data_dir, cache_dir=cache_dir, wtg_numbers=[1]) + + # wraps= keeps the real unpack behaviour; the spy only records whether it was called. + with patch.object(hot, "_unpack_hot_10min_year", wraps=hot._unpack_hot_10min_year) as spy: # noqa: SLF001 + _load(data_dir=data_dir, cache_dir=cache_dir, wtg_numbers=[1]) + + assert spy.call_count == 0 + + +def test_growing_subset_only_unpacks_the_new_turbine(tmp_path: Path) -> None: + """Adding T03 to an existing T01 cache unpacks only T03 and reuses T01.""" + data_dir, cache_dir = _dirs(tmp_path) + _load(data_dir=data_dir, cache_dir=cache_dir, wtg_numbers=[1]) + + with patch.object(hot, "_unpack_hot_10min_year", wraps=hot._unpack_hot_10min_year) as spy: # noqa: SLF001 + wide = _load(data_dir=data_dir, cache_dir=cache_dir, wtg_numbers=[1, 3]) + + spy.assert_called_once() + assert list(spy.call_args.kwargs["serial_numbers"]) == [hot._HOT_SERIAL_OFFSET + 3] # noqa: SLF001 only the new + assert list(cache_dir.glob("hot10min_2020_T01_*.parquet")) + assert list(cache_dir.glob("hot10min_2020_T03_*.parquet")) + assert {col[0] for col in wide.columns} == {"T01", "T03"} diff --git a/tests/benchmarking/synthetic/test_cp_core.py b/tests/benchmarking/synthetic/test_cp_core.py new file mode 100644 index 00000000..0196140d --- /dev/null +++ b/tests/benchmarking/synthetic/test_cp_core.py @@ -0,0 +1,164 @@ +"""Tests for the analytic Hill of Towie Cp surface and Cp-space physics core.""" + +from __future__ import annotations + +import numpy as np +import pytest + +from benchmarking.synthetic.cp_core import ( + HOT_CP_MODEL, + CpCore, + cp_surface, + power_from_cp_change, + region2_fraction, + rpm_from_power, + rpm_from_power_change, +) + + +def test_cp_peaks_at_optimal_tsr_and_pitch() -> None: + """At optimal TSR and pitch the surface returns Cp_max.""" + cp = cp_surface(tsr=HOT_CP_MODEL.opt_tsr, pitch=HOT_CP_MODEL.opt_pitch, params=HOT_CP_MODEL) + assert cp == HOT_CP_MODEL.cp_max + + +def test_cp_off_optimal_pitch_matches_hand_calculation() -> None: + """At optimal TSR the TSR factor is exactly 1, so Cp = cp_max * (1 - pitch_scale * opt_pitch^2).""" + cp = cp_surface(tsr=HOT_CP_MODEL.opt_tsr, pitch=0.0, params=HOT_CP_MODEL) + expected = HOT_CP_MODEL.cp_max * (1.0 - HOT_CP_MODEL.pitch_scale * HOT_CP_MODEL.opt_pitch**2) + assert cp == pytest.approx(expected) + + +def test_cp_surface_is_vectorised() -> None: + """Array inputs return an array of the per-element scalar Cp values.""" + cp = cp_surface(tsr=[7.0, 8.0], pitch=[-1.0, 0.0], params=HOT_CP_MODEL) + assert cp.shape == (2,) + expected = [ + float(cp_surface(tsr=7.0, pitch=-1.0, params=HOT_CP_MODEL)), + float(cp_surface(tsr=8.0, pitch=0.0, params=HOT_CP_MODEL)), + ] + np.testing.assert_allclose(cp, expected) + + +def test_cp_clamps_to_zero_far_from_peak() -> None: + """Far from the optimal TSR the quadratic falloff is clamped to zero, never negative.""" + cp = cp_surface(tsr=20.0, pitch=HOT_CP_MODEL.opt_pitch, params=HOT_CP_MODEL) + assert cp == 0.0 + + +def test_region2_fraction_is_quarter_at_sigmoid_midpoint() -> None: + """At the sigmoid midpoint (2000 kW) the squared sigmoid is exactly 0.25.""" + assert region2_fraction(2000.0) == pytest.approx(0.25) + + +def test_region2_fraction_near_one_deep_in_region2() -> None: + """Deep in region 2 (low power) almost all of the period is region 2.""" + assert region2_fraction(500.0) > 0.99 + + +def test_region2_fraction_tails_off_near_rated() -> None: + """Approaching rated power (2300 kW) the region-2 fraction tails toward zero.""" + assert region2_fraction(2300.0) < 0.01 + + +def test_region2_fraction_decreases_monotonically_with_power() -> None: + """The region-2 fraction is monotonically non-increasing in power.""" + powers = np.arange(0.0, 2400.0, 50.0) + fractions = region2_fraction(powers) + assert np.all(np.diff(fractions) <= 0.0) + + +def test_power_from_cp_change_applies_region2_weighted_ratio() -> None: + """At 2000 kW (region-2 fraction 0.25) a +10% Cp gives +2.5% power: 2000 -> 2050.""" + new = power_from_cp_change(2000.0, cp_ratio=1.10, rated_power_kw=2300.0) + assert new == pytest.approx(2050.0) + + +def test_power_from_cp_change_clips_at_rated() -> None: + """A large Cp ratio below the near-rated band cannot push power above the rated clip.""" + new = power_from_cp_change(2200.0, cp_ratio=5.0, rated_power_kw=2300.0) + assert new == pytest.approx(2300.0) + + +def test_power_from_cp_change_leaves_near_rated_power_unchanged() -> None: + """A virtually-pure-rated record is untouched by a Cp change (positive or negative).""" + assert power_from_cp_change(2295.0, cp_ratio=1.03, rated_power_kw=2300.0) == pytest.approx(2295.0) + assert power_from_cp_change(2295.0, cp_ratio=0.97, rated_power_kw=2300.0) == pytest.approx(2295.0) + + +def test_power_from_cp_change_reduces_power_for_negative_delta() -> None: + """A Cp ratio below 1 reduces power roughly proportionally deep in region 2.""" + new = power_from_cp_change(800.0, cp_ratio=0.90, rated_power_kw=2300.0) + assert new == pytest.approx(800.0 * 0.90, rel=1e-3) + + +def test_power_from_cp_change_leaves_negative_power_unchanged() -> None: + """A non-producing (negative power) record is untouched by a positive Cp change.""" + new = power_from_cp_change(-5.0, cp_ratio=1.03, rated_power_kw=2300.0) + assert new == pytest.approx(-5.0) + + +def test_power_from_cp_change_leaves_negative_power_unchanged_for_negative_delta() -> None: + """A degraded-blades (negative) Cp change is still a no-op on non-producing records.""" + new = power_from_cp_change(-5.0, cp_ratio=0.90, rated_power_kw=2300.0) + assert new == pytest.approx(-5.0) + + +def test_power_from_cp_change_leaves_zero_power_unchanged() -> None: + """A zero-power record is left at zero by a Cp change.""" + new = power_from_cp_change(0.0, cp_ratio=1.10, rated_power_kw=2300.0) + assert new == pytest.approx(0.0) + + +def test_power_from_cp_change_is_vectorised() -> None: + """Array inputs return an array of modified powers; non-producing rows are untouched.""" + new = power_from_cp_change(np.array([-5.0, 0.0, 2000.0]), cp_ratio=1.10, rated_power_kw=2300.0) + assert new.shape == (3,) + assert new[0] == pytest.approx(-5.0) + assert new[1] == pytest.approx(0.0) + assert new[2] == pytest.approx(2050.0) + + +def test_rpm_from_power_matches_curve_anchor() -> None: + """The ported rpm-vs-power curve passes through its known knot (1000 kW -> 1392 rpm).""" + assert rpm_from_power(1000.0) == pytest.approx(1392.0) + + +def test_rpm_unchanged_when_power_unchanged() -> None: + """If power does not change, rpm is unchanged.""" + new_rpm = rpm_from_power_change(baseline_rpm=1392.0, baseline_power_kw=1000.0, new_power_kw=1000.0) + assert new_rpm == pytest.approx(1392.0) + + +def test_rpm_increases_with_power() -> None: + """Higher power scales rpm up by the curve ratio.""" + new_rpm = rpm_from_power_change(baseline_rpm=1392.0, baseline_power_kw=1000.0, new_power_kw=1100.0) + assert new_rpm == pytest.approx(rpm_from_power(1100.0)) + assert new_rpm > 1392.0 + + +def test_rpm_unchanged_when_new_power_not_positive() -> None: + """A non-positive new power leaves rpm at its baseline value.""" + new_rpm = rpm_from_power_change(baseline_rpm=1392.0, baseline_power_kw=1000.0, new_power_kw=0.0) + assert new_rpm == pytest.approx(1392.0) + + +def test_cpcore_defaults_to_hot_model() -> None: + """A default CpCore uses the HoT Cp model and 2300 kW rated power.""" + core = CpCore() + assert core.cp_params == HOT_CP_MODEL + assert core.rated_power_kw == pytest.approx(2300.0) + + +def test_cpcore_apply_cp_ratio_uses_its_rated_power() -> None: + """CpCore.apply_cp_ratio clips at the core's configured rated power.""" + core = CpCore(rated_power_kw=2300.0) + assert core.apply_cp_ratio(2000.0, cp_ratio=1.10) == pytest.approx(2050.0) + assert core.apply_cp_ratio(2200.0, cp_ratio=5.0) == pytest.approx(2300.0) + + +def test_cpcore_rpm_after_tracks_power() -> None: + """CpCore.rpm_after scales rpm by the operating-curve ratio.""" + core = CpCore() + new_rpm = core.rpm_after(baseline_rpm=1392.0, baseline_power_kw=1000.0, new_power_kw=1100.0) + assert new_rpm == pytest.approx(rpm_from_power(1100.0)) diff --git a/tests/benchmarking/synthetic/test_generator.py b/tests/benchmarking/synthetic/test_generator.py new file mode 100644 index 00000000..ff9f8602 --- /dev/null +++ b/tests/benchmarking/synthetic/test_generator.py @@ -0,0 +1,216 @@ +"""Tests for the synthetic dataset generator.""" + +from __future__ import annotations + +import json +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.generator import ( + SyntheticDataset, + ToggleSchedule, + _treated_mask, + generate_dataset, + treated_mask, +) +from benchmarking.synthetic.upgrades import ConstantCpChange +from wind_up.constants import TIMESTAMP_COL + +if TYPE_CHECKING: + from pathlib import Path + + +def _wf_df( + *, + turbines: tuple[str, ...] = ("T01", "T02"), + start: str = "2020-01-01", + periods: int = 288, + power: float = 1000.0, +) -> pd.DataFrame: + index = pd.date_range(start, periods=periods, freq="10min", tz="UTC") + frames = [] + for turbine in turbines: + frame = pd.DataFrame( + { + HOT_COLUMNS.turbine: turbine, + HOT_COLUMNS.active_power: float(power), + HOT_COLUMNS.wind_speed: 8.0, + HOT_COLUMNS.wind_speed_sd: 0.8, + HOT_COLUMNS.gen_rpm: 1400.0, + }, + index=index, + ) + frames.append(frame) + wf_df = pd.concat(frames) + wf_df.index.name = TIMESTAMP_COL + return wf_df + + +def test_prepost_modifies_only_post_rows_of_test_turbine() -> None: + """In prepost mode only the test turbine's post-changeover power changes.""" + wf_df = _wf_df() + timestamps = wf_df.index.unique() + changeover = timestamps[len(timestamps) // 2] + + dataset = generate_dataset( + scada_df=wf_df, + test_wtgs=["T01"], + upgrades=[ConstantCpChange(delta=0.10)], + mode="prepost", + upgrade_timing=changeover, + ) + synthetic = dataset.synthetic_df + + t01 = synthetic[synthetic[HOT_COLUMNS.turbine] == "T01"] + pre = t01[t01.index < changeover][HOT_COLUMNS.active_power] + post = t01[t01.index >= changeover][HOT_COLUMNS.active_power] + assert np.allclose(pre.to_numpy(), 1000.0) + assert np.all(post.to_numpy() > 1000.0) + + t02 = synthetic[synthetic[HOT_COLUMNS.turbine] == "T02"][HOT_COLUMNS.active_power] + assert np.allclose(t02.to_numpy(), 1000.0) + + +def test_generator_retains_unchanged_original() -> None: + """The returned original_df equals the input SCADA and is not mutated.""" + wf_df = _wf_df() + timestamps = wf_df.index.unique() + dataset = generate_dataset( + scada_df=wf_df, + test_wtgs=["T01"], + upgrades=[ConstantCpChange(delta=0.10)], + mode="prepost", + upgrade_timing=timestamps[len(timestamps) // 2], + ) + pd.testing.assert_frame_equal(dataset.original_df, wf_df) + + +def test_toggle_modifies_alternate_blocks() -> None: + """In toggle mode ``period`` is a full on/off cycle: half off then half on.""" + wf_df = _wf_df(periods=288) # two days at 10-min + dataset = generate_dataset( + scada_df=wf_df, + test_wtgs=["T01"], + upgrades=[ConstantCpChange(delta=0.10)], + mode="toggle", + upgrade_timing=ToggleSchedule(period=pd.Timedelta(days=1)), + ) + t01 = dataset.synthetic_df[dataset.synthetic_df[HOT_COLUMNS.turbine] == "T01"] + start = t01.index.min() + half = pd.Timedelta(hours=12) + first_half = t01[t01.index < start + half][HOT_COLUMNS.active_power] + second_half = t01[(t01.index >= start + half) & (t01.index < start + 2 * half)][HOT_COLUMNS.active_power] + assert np.allclose(first_half.to_numpy(), 1000.0) # first half-period: toggle off + assert np.all(second_half.to_numpy() > 1000.0) # second half-period: toggle on + + +def test_toggle_start_leaves_pre_start_rows_untreated() -> None: + """A ToggleSchedule.start places baseline before toggling: pre-start rows stay untreated.""" + wf_df = _wf_df(periods=288) # two days at 10-min + timestamps = wf_df.index.unique() + start = timestamps[len(timestamps) // 2] # toggling begins at the midpoint + + dataset = generate_dataset( + scada_df=wf_df, + test_wtgs=["T01"], + upgrades=[ConstantCpChange(delta=0.10)], + mode="toggle", + upgrade_timing=ToggleSchedule(period=pd.Timedelta(hours=12), start=start), + ) + t01 = dataset.synthetic_df[dataset.synthetic_df[HOT_COLUMNS.turbine] == "T01"] + before = t01[t01.index < start][HOT_COLUMNS.active_power] + assert np.allclose(before.to_numpy(), 1000.0) # baseline before toggling: untreated + + # start_on defaults False: [start, start+6h) off, [start+6h, start+12h) on. + half = pd.Timedelta(hours=6) + off_block = t01[(t01.index >= start) & (t01.index < start + half)][HOT_COLUMNS.active_power] + on_block = t01[(t01.index >= start + half) & (t01.index < start + 2 * half)][HOT_COLUMNS.active_power] + assert np.allclose(off_block.to_numpy(), 1000.0) + assert np.all(on_block.to_numpy() > 1000.0) + + +def test_public_treated_mask_infers_mode_from_timing_type() -> None: + """treated_mask infers prepost vs toggle from the upgrade_timing type.""" + wf_df = _wf_df(turbines=("T01",), periods=288) + index = wf_df.index + changeover = index[len(index) // 2] + + prepost_mask = treated_mask(index, changeover) + assert np.array_equal(prepost_mask, np.asarray(index >= changeover)) + + schedule = ToggleSchedule(period=pd.Timedelta(hours=12), start=changeover) + toggle_mask = treated_mask(index, schedule) + assert not toggle_mask[np.asarray(index < changeover)].any() # baseline untreated + assert toggle_mask[np.asarray(index >= changeover)].any() # some on-rows after start + + +def test_treated_mask_rejects_mode_timing_mismatch() -> None: + """A mode that disagrees with the upgrade_timing type fails fast with a clear TypeError.""" + index = _wf_df(turbines=("T01",), periods=48).index + changeover = index[len(index) // 2] + schedule = ToggleSchedule(period=pd.Timedelta(hours=12), start=changeover) + + with pytest.raises(TypeError, match="prepost mode needs a changeover Timestamp"): + _treated_mask(index, mode="prepost", upgrade_timing=schedule) + with pytest.raises(TypeError, match="toggle mode needs a ToggleSchedule"): + _treated_mask(index, mode="toggle", upgrade_timing=changeover) + + +def test_run_metadata_records_recipe() -> None: + """Run metadata captures the test turbines, mode, seed and upgrade descriptions.""" + wf_df = _wf_df() + timestamps = wf_df.index.unique() + dataset = generate_dataset( + scada_df=wf_df, + test_wtgs=["T01"], + upgrades=[ConstantCpChange(delta=0.10)], + mode="prepost", + upgrade_timing=timestamps[len(timestamps) // 2], + seed=7, + ) + metadata = dataset.run_metadata + assert metadata["test_wtgs"] == ["T01"] + assert metadata["mode"] == "prepost" + assert metadata["seed"] == 7 + assert metadata["upgrades"][0]["kind"] == "constant_cp" + + +def _prepost_constant_dataset(delta: float = 0.10) -> SyntheticDataset: + wf_df = _wf_df() + timestamps = wf_df.index.unique() + return generate_dataset( + scada_df=wf_df, + test_wtgs=["T01"], + upgrades=[ConstantCpChange(delta=delta)], + mode="prepost", + upgrade_timing=timestamps[len(timestamps) // 2], + ) + + +def test_dataset_true_uplift_recovers_injected_effect() -> None: + """SyntheticDataset.true_uplift recovers the region-2-weighted injected uplift.""" + dataset = _prepost_constant_dataset(delta=0.10) + result = dataset.true_uplift() + # constant power 1000 kW, region-2 fraction ~0.999 -> ~+9.99% + assert result.overall == pytest.approx(0.0999, rel=1e-3) + + +def test_save_writes_roundtrippable_files(tmp_path: Path) -> None: + """save() writes synthetic, original and metadata (with ground truth) that round-trip.""" + dataset = _prepost_constant_dataset() + dataset.save(tmp_path) + + assert (tmp_path / "synthetic.parquet").exists() + assert (tmp_path / "original.parquet").exists() + assert (tmp_path / "run_metadata.json").exists() + + roundtrip = pd.read_parquet(tmp_path / "synthetic.parquet") + pd.testing.assert_frame_equal(roundtrip, dataset.synthetic_df) + + metadata = json.loads((tmp_path / "run_metadata.json").read_text()) + assert "ground_truth" in metadata + assert metadata["ground_truth"]["T01"]["overall"] == pytest.approx(0.0999, rel=1e-3) diff --git a/tests/benchmarking/synthetic/test_ground_truth.py b/tests/benchmarking/synthetic/test_ground_truth.py new file mode 100644 index 00000000..b4597217 --- /dev/null +++ b/tests/benchmarking/synthetic/test_ground_truth.py @@ -0,0 +1,114 @@ +"""Tests for comparison-derived ground-truth uplift.""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.ground_truth import true_uplift +from wind_up.constants import TIMESTAMP_COL + + +def _paired( + orig_power: list[float], + syn_power: list[float], + *, + wtg: str = "T01", + ws: list[float] | None = None, +) -> tuple[pd.DataFrame, pd.DataFrame]: + n = len(orig_power) + index = pd.date_range("2020-01-01", periods=n, freq="10min", tz="UTC") + + def frame(power: list[float]) -> pd.DataFrame: + df = pd.DataFrame( + { + HOT_COLUMNS.turbine: wtg, + HOT_COLUMNS.active_power: np.array(power, dtype=float), + HOT_COLUMNS.wind_speed: np.array(ws if ws is not None else [8.0] * n, dtype=float), + HOT_COLUMNS.wind_speed_sd: 0.8, + }, + index=index, + ) + df.index.name = TIMESTAMP_COL + return df + + return frame(orig_power), frame(syn_power) + + +def test_overall_uplift_is_energy_ratio_of_synthetic_to_original() -> None: + """Overall uplift is sum(synthetic)/sum(original) - 1 over the treated records.""" + original, synthetic = _paired([1000.0] * 10, [1050.0] * 10) + result = true_uplift(synthetic, original, test_wtg="T01") + assert result.overall == pytest.approx(0.05) + + +def test_default_mask_is_changed_records_only() -> None: + """With no mask, untreated (unchanged) records are excluded from the ratio.""" + # first two rows unchanged, last two +10% + original, synthetic = _paired([1000.0, 1000.0, 1000.0, 1000.0], [1000.0, 1000.0, 1100.0, 1100.0]) + result = true_uplift(synthetic, original, test_wtg="T01") + assert result.overall == pytest.approx(0.10) + + +def test_uplift_depends_on_record_window() -> None: + """Passing an explicit mask covering all records dilutes the uplift (record-dependent).""" + original, synthetic = _paired([1000.0, 1000.0, 1000.0, 1000.0], [1000.0, 1000.0, 1100.0, 1100.0]) + full = true_uplift(synthetic, original, test_wtg="T01", mask=np.array([True, True, True, True])) + assert full.overall == pytest.approx(0.05) + + +def test_overall_uplift_ignores_nan_power_records() -> None: + """Real SCADA can have NaN power; those records must not poison the ratio. + + Rows where the original power is NaN are unchanged by the upgrade (NaN stays NaN), + so they must be excluded rather than (a) flagged as treated via ``NaN != NaN`` or + (b) turning the energy sums into NaN. + """ + original, synthetic = _paired( + [1000.0, np.nan, 1000.0, np.nan], + [1100.0, np.nan, 1100.0, np.nan], + ) + result = true_uplift(synthetic, original, test_wtg="T01") + assert result.overall == pytest.approx(0.10) + + +def test_explicit_mask_drops_nan_power_records() -> None: + """An explicit mask still excludes NaN-power rows from the energy ratio.""" + original, synthetic = _paired( + [1000.0, np.nan, 1000.0, 1000.0], + [1100.0, np.nan, 1100.0, 1000.0], + ) + result = true_uplift(synthetic, original, test_wtg="T01", mask=np.array([True, True, True, True])) + # finite rows: (1100+1100+1000)/(1000+1000+1000) - 1 + assert result.overall == pytest.approx((1100.0 + 1100.0 + 1000.0) / 3000.0 - 1.0) + + +def test_per_condition_breakdown_by_wind_speed() -> None: + """A by='ws' breakdown reports the true uplift within original wind-speed bins.""" + original, synthetic = _paired( + [800.0, 800.0, 1500.0, 1500.0], + [880.0, 880.0, 1530.0, 1530.0], + ws=[6.0, 6.0, 10.0, 10.0], + ) + result = true_uplift(synthetic, original, test_wtg="T01", by="ws", bins=[5.0, 8.0, 12.0]) + by_condition = result.by_condition + assert by_condition is not None + assert len(by_condition) == 2 + np.testing.assert_allclose(by_condition["true_uplift"].to_numpy(), [0.10, 0.02]) + + +def test_per_condition_breakdown_by_power_uses_original_untreated_power() -> None: + """A by='power' breakdown bins on the ORIGINAL (untreated) power, not the treated synthetic power. + + The first record's original power (990) sits below the 1000 edge while its synthetic power (1010) + sits above it: binning on original keeps it in the low bin, so that bin is populated (a synthetic- + power binning would leave it empty). + """ + original, synthetic = _paired([990.0, 1800.0], [1010.0, 1836.0]) + result = true_uplift(synthetic, original, test_wtg="T01", by="power", bins=[0.0, 1000.0, 2000.0]) + by_condition = result.by_condition + assert by_condition is not None + assert by_condition["n_records"].to_numpy().tolist() == [1, 1] + np.testing.assert_allclose(by_condition["true_uplift"].to_numpy(), [1010.0 / 990.0 - 1.0, 0.02]) diff --git a/tests/benchmarking/synthetic/test_make_example_datasets.py b/tests/benchmarking/synthetic/test_make_example_datasets.py new file mode 100644 index 00000000..683139a6 --- /dev/null +++ b/tests/benchmarking/synthetic/test_make_example_datasets.py @@ -0,0 +1,105 @@ +"""Tests for the example-dataset driver (one dataset per Issue 1 profile).""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import numpy as np +import pandas as pd +import pytest + +import benchmarking.synthetic.make_example_datasets as driver +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.make_example_datasets import example_profiles, generate_example_datasets, main +from wind_up.constants import TIMESTAMP_COL + +if TYPE_CHECKING: + from pathlib import Path + +EXPECTED_PROFILES = {"constant_cp", "wind_speed_cp", "ti_cp", "rated_power"} + + +def _swept_wf_df( + *, periods: int = 288, turbines: tuple[str, ...] = ("T01", "T02"), start: str = "2020-01-01" +) -> pd.DataFrame: + index = pd.date_range(start, periods=periods, freq="10min", tz="UTC") + ws = np.linspace(3.0, 14.0, periods) + power = np.clip(2300.0 * ((ws - 3.0) / 11.0) ** 3, 0.0, 2300.0) + frames = [ + pd.DataFrame( + { + HOT_COLUMNS.turbine: turbine, + HOT_COLUMNS.active_power: power, + HOT_COLUMNS.wind_speed: ws, + HOT_COLUMNS.wind_speed_sd: 0.1 * ws, + HOT_COLUMNS.gen_rpm: 1400.0, + }, + index=index, + ) + for turbine in turbines + ] + wf_df = pd.concat(frames) + wf_df.index.name = TIMESTAMP_COL + return wf_df + + +def test_example_profiles_cover_the_four_issue1_profiles() -> None: + """The driver defines the four Issue 1 profiles, each a non-empty upgrade list.""" + profiles = example_profiles() + assert set(profiles) == EXPECTED_PROFILES + assert all(len(upgrades) >= 1 for upgrades in profiles.values()) + + +def test_generate_example_datasets_writes_one_dataset_per_profile(tmp_path: Path) -> None: + """Each profile produces a saved dataset with a non-zero injected uplift.""" + wf_df = _swept_wf_df() + timestamps = wf_df.index.unique() + datasets = generate_example_datasets( + scada_df=wf_df, + test_wtgs=["T01"], + mode="prepost", + upgrade_timing=timestamps[len(timestamps) // 2], + out_root=tmp_path, + ) + + assert set(datasets) == EXPECTED_PROFILES + for name, dataset in datasets.items(): + assert (tmp_path / name / "synthetic.parquet").exists() + assert (tmp_path / name / "run_metadata.json").exists() + assert (tmp_path / name / "power_curve_T01.png").exists() + assert dataset.true_uplift().overall != 0.0 + + +def test_main_wires_hot_loader_to_the_driver(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + """``main`` loads SCADA via ``load_hot_scada`` and writes one dataset per profile.""" + captured: dict = {} + + def fake_load_hot_scada(**kwargs: object) -> tuple[pd.DataFrame, pd.DataFrame]: + captured.update(kwargs) + # Span exactly the requested window, as the real loader does, so the + # mid-window changeover lands on real rows. + start_dt = kwargs["start_dt"] + end_dt_excl = kwargs["end_dt_excl"] + periods = int((end_dt_excl - start_dt) / pd.Timedelta(minutes=10)) + wf_df = _swept_wf_df(periods=periods, start=str(start_dt)) + return wf_df, pd.DataFrame({"Name": ["T01", "T02"]}) + + monkeypatch.setattr(driver, "load_hot_scada", fake_load_hot_scada) + + datasets = main(out_root=tmp_path, test_wtg="T01") + + assert set(datasets) == EXPECTED_PROFILES + for name in EXPECTED_PROFILES: + assert (tmp_path / name / "synthetic.parquet").exists() + # The injection changeover must land inside the loaded window so rows are treated. + assert captured["start_dt"] < captured["end_dt_excl"] + + +@pytest.mark.slow +def test_main_end_to_end_downloads_real_hot_data(tmp_path: Path) -> None: + """End-to-end: download real Hill of Towie data and produce datasets (network).""" + datasets = main(out_root=tmp_path / "out", data_dir=tmp_path / "data", test_wtg="T01") + assert set(datasets) == EXPECTED_PROFILES + for name in EXPECTED_PROFILES: + assert (tmp_path / "out" / name / "synthetic.parquet").exists() + assert datasets[name].true_uplift().overall != 0.0 diff --git a/tests/benchmarking/synthetic/test_plots.py b/tests/benchmarking/synthetic/test_plots.py new file mode 100644 index 00000000..393bdf40 --- /dev/null +++ b/tests/benchmarking/synthetic/test_plots.py @@ -0,0 +1,122 @@ +"""Tests for the synthetic-dataset verification plots.""" + +from __future__ import annotations + +from typing import TYPE_CHECKING + +import matplotlib as mpl + +mpl.use("Agg") # headless: no display needed for tests + +import numpy as np +import pandas as pd + +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.generator import generate_dataset +from benchmarking.synthetic.plots import plot_power_curve_comparison +from benchmarking.synthetic.upgrades import ConstantCpChange +from wind_up.constants import TIMESTAMP_COL + +if TYPE_CHECKING: + from pathlib import Path + + +def _swept_dataset() -> object: + periods = 288 + index = pd.date_range("2020-01-01", periods=periods, freq="10min", tz="UTC") + ws = np.linspace(3.0, 14.0, periods) + power = np.clip(2300.0 * ((ws - 3.0) / 11.0) ** 3, 0.0, 2300.0) + frames = [ + pd.DataFrame( + { + HOT_COLUMNS.turbine: turbine, + HOT_COLUMNS.active_power: power, + HOT_COLUMNS.wind_speed: ws, + HOT_COLUMNS.wind_speed_sd: 0.1 * ws, + HOT_COLUMNS.gen_rpm: 1400.0, + }, + index=index, + ) + for turbine in ("T01", "T02") + ] + wf_df = pd.concat(frames) + wf_df.index.name = TIMESTAMP_COL + return generate_dataset( + scada_df=wf_df, + test_wtgs=["T01"], + upgrades=[ConstantCpChange(delta=0.05)], + mode="prepost", + upgrade_timing=index[periods // 2], + ) + + +def _paired_dfs(orig_power: list[float], syn_power: list[float], ws: list[float]) -> tuple[pd.DataFrame, pd.DataFrame]: + n = len(orig_power) + index = pd.date_range("2020-01-01", periods=n, freq="10min", tz="UTC") + + def frame(power: list[float]) -> pd.DataFrame: + df = pd.DataFrame( + { + HOT_COLUMNS.turbine: "T01", + HOT_COLUMNS.active_power: np.array(power, dtype=float), + HOT_COLUMNS.wind_speed: np.array(ws, dtype=float), + }, + index=index, + ) + df.index.name = TIMESTAMP_COL + return df + + return frame(syn_power), frame(orig_power) + + +def test_power_curve_comparison_has_three_panels_with_aligned_axes() -> None: + """Three panels; the two power-curve panels share x and y, all three share x.""" + dataset = _swept_dataset() + fig = plot_power_curve_comparison(dataset.synthetic_df, dataset.original_df, test_wtg="T01") + + assert len(fig.axes) == 3 + ax_orig, ax_syn, ax_delta = fig.axes + assert ax_orig.get_xlim() == ax_syn.get_xlim() == ax_delta.get_xlim() + assert ax_orig.get_ylim() == ax_syn.get_ylim() + + +def test_third_panel_shows_kw_change() -> None: + """The third panel plots the synthetic-minus-original power change.""" + dataset = _swept_dataset() + fig = plot_power_curve_comparison(dataset.synthetic_df, dataset.original_df, test_wtg="T01") + ax_delta = fig.axes[2] + assert "change" in ax_delta.get_ylabel().lower() + assert any(coll.get_offsets().size > 0 for coll in ax_delta.collections) + + +def test_power_curve_comparison_enables_grid() -> None: + """All panels show gridlines (per the verification request).""" + dataset = _swept_dataset() + fig = plot_power_curve_comparison(dataset.synthetic_df, dataset.original_df, test_wtg="T01") + for ax in fig.axes: + assert any(line.get_visible() for line in ax.get_xgridlines()) + assert any(line.get_visible() for line in ax.get_ygridlines()) + + +def test_kw_change_panel_excludes_nan_downtime_rows() -> None: + """NaN-power (downtime) rows are not treated as changed in the kW-change panel.""" + # row 0 changed; row 1 both-NaN (downtime); row 2 unchanged; row 3 changed + synthetic, original = _paired_dfs( + orig_power=[1000.0, np.nan, 1000.0, 1000.0], + syn_power=[1100.0, np.nan, 1000.0, 1100.0], + ws=[8.0, 9.0, 10.0, 11.0], + ) + fig = plot_power_curve_comparison(synthetic, original, test_wtg="T01") + ax_delta = fig.axes[2] + plotted = np.concatenate([coll.get_offsets().data for coll in ax_delta.collections if coll.get_offsets().size]) + # exactly the two genuinely changed finite rows (ws 8 and 11), not the NaN row + assert sorted(plotted[:, 0].tolist()) == [8.0, 11.0] + + +def test_power_curve_comparison_saves_file(tmp_path: Path) -> None: + """A PNG is written when ``save_path`` is given.""" + dataset = _swept_dataset() + save_path = tmp_path / "power_curve_comparison.png" + plot_power_curve_comparison(dataset.synthetic_df, dataset.original_df, test_wtg="T01", save_path=save_path) + assert save_path.exists() + assert save_path.stat().st_size > 0 diff --git a/tests/benchmarking/synthetic/test_public_api.py b/tests/benchmarking/synthetic/test_public_api.py new file mode 100644 index 00000000..1fa822b2 --- /dev/null +++ b/tests/benchmarking/synthetic/test_public_api.py @@ -0,0 +1,22 @@ +"""The synthetic package exposes its main entry points at the package root.""" + +from __future__ import annotations + +import benchmarking.synthetic as synth + + +def test_public_api_exports_core_entry_points() -> None: + for name in ( + "generate_dataset", + "SyntheticDataset", + "ToggleSchedule", + "treated_mask", + "true_uplift", + "CpCore", + "HOT_CP_MODEL", + "ConstantCpChange", + "WindSpeedCpChange", + "ConditionCpChange", + "RatedPowerChange", + ): + assert hasattr(synth, name), name diff --git a/tests/benchmarking/synthetic/test_upgrades.py b/tests/benchmarking/synthetic/test_upgrades.py new file mode 100644 index 00000000..d8740a84 --- /dev/null +++ b/tests/benchmarking/synthetic/test_upgrades.py @@ -0,0 +1,139 @@ +"""Tests for the synthetic upgrade callables and their resolution.""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from benchmarking.synthetic import HOT_COLUMNS +from benchmarking.synthetic.cp_core import CpCore, region2_fraction +from benchmarking.synthetic.upgrades import ( + ConditionCpChange, + ConstantCpChange, + RatedPowerChange, + WindSpeedCpChange, + apply_upgrades, +) + + +def _rows( + powers: list[float], + *, + ws: list[float] | None = None, + sd: list[float] | None = None, + rpm: list[float] | None = None, +) -> pd.DataFrame: + n = len(powers) + return pd.DataFrame( + { + HOT_COLUMNS.active_power: np.array(powers, dtype=float), + HOT_COLUMNS.wind_speed: np.array(ws if ws is not None else [8.0] * n, dtype=float), + HOT_COLUMNS.wind_speed_sd: np.array(sd if sd is not None else [0.8] * n, dtype=float), + HOT_COLUMNS.gen_rpm: np.array(rpm if rpm is not None else [1400.0] * n, dtype=float), + } + ) + + +def test_constant_cp_change_scales_region2_power() -> None: + """At 2000 kW (region-2 fraction 0.25) a +10% Cp gives 2050 kW (not 2200 due to 25% region 2).""" + rows = _rows([2000.0]) + out = apply_upgrades(rows, [ConstantCpChange(delta=0.10)], cp=CpCore(rated_power_kw=2300.0)) + assert out[HOT_COLUMNS.active_power].iloc[0] == pytest.approx(2050.0) + + +def test_empty_upgrade_list_leaves_rows_unchanged() -> None: + """Applying no upgrades returns power identical to the original.""" + rows = _rows([500.0, 1500.0, 2200.0]) + out = apply_upgrades(rows, [], cp=CpCore()) + pd.testing.assert_series_equal(out[HOT_COLUMNS.active_power], rows[HOT_COLUMNS.active_power]) + + +def test_constant_cp_change_only_treats_producing_below_rated_records() -> None: + """Non-producing (<=0) and virtually-rated rows are not treated; region-2 rows are. + + Locks the user-visible behaviour: a Cp change only scales genuine producing records + below pure rated, so idling/curtailed and at-rated records keep their original power. + """ + rows = _rows([-5.0, 0.0, 1000.0, 2295.0]) + out = apply_upgrades(rows, [ConstantCpChange(delta=0.03)], cp=CpCore(rated_power_kw=2300.0)) + power = out[HOT_COLUMNS.active_power].to_numpy() + assert power[0] == pytest.approx(-5.0) # negative: untouched + assert power[1] == pytest.approx(0.0) # zero: untouched + assert power[2] > 1000.0 # producing region 2: still changed + assert power[3] == pytest.approx(2295.0) # virtually rated: untouched + + +def test_constant_cp_change_does_not_mutate_input() -> None: + """apply_upgrades returns a new frame and leaves the caller's rows untouched.""" + rows = _rows([1500.0]) + original = rows[HOT_COLUMNS.active_power].copy() + apply_upgrades(rows, [ConstantCpChange(delta=0.05)], cp=CpCore()) + pd.testing.assert_series_equal(rows[HOT_COLUMNS.active_power], original) + + +def test_wind_speed_cp_change_applies_delta_by_original_wind_speed() -> None: + """Cp delta is interpolated over the original wind speed; zero-delta bins are unchanged.""" + rows = _rows([600.0, 1000.0, 1500.0], ws=[4.0, 8.0, 12.0]) + upgrade = WindSpeedCpChange(ws_points=[4.0, 8.0, 12.0], deltas=[0.0, 0.04, 0.0]) + out = apply_upgrades(rows, [upgrade], cp=CpCore()) + power = out[HOT_COLUMNS.active_power].to_numpy() + f8 = region2_fraction(1000.0) + assert power[0] == pytest.approx(600.0) # ws 4: zero delta + assert power[1] == pytest.approx(1000.0 * (1.0 + f8 * 0.04)) # ws 8: +4% Cp in region 2 + assert power[2] == pytest.approx(1500.0) # ws 12: zero delta + + +def test_condition_cp_change_varies_by_original_turbulence_intensity() -> None: + """Cp delta is interpolated over original TI = WindSpeedSD / WindSpeedMean.""" + # ws 8 m/s with sd 0.8 and 1.6 -> TI 0.10 and 0.20 + rows = _rows([1000.0, 1000.0], ws=[8.0, 8.0], sd=[0.8, 1.6]) + upgrade = ConditionCpChange(by="ti", points=[0.10, 0.20], deltas=[0.0, 0.05]) + out = apply_upgrades(rows, [upgrade], cp=CpCore()) + power = out[HOT_COLUMNS.active_power].to_numpy() + f = region2_fraction(1000.0) + assert power[0] == pytest.approx(1000.0) # TI 0.10: zero delta + assert power[1] == pytest.approx(1000.0 * (1.0 + f * 0.05)) # TI 0.20: +5% Cp in region 2 + + +def test_rated_power_downrate_clips_at_new_rated() -> None: + """A downrate caps power at the new rated and leaves region-2 power unchanged.""" + rows = _rows([1000.0, 2200.0]) + out = apply_upgrades(rows, [RatedPowerChange(new_rated_power_kw=2000.0)], cp=CpCore(rated_power_kw=2300.0)) + power = out[HOT_COLUMNS.active_power].to_numpy() + assert power[0] == pytest.approx(1000.0) # well below new rated: unchanged + assert power[1] == pytest.approx(2000.0) # above new rated: clipped + + +def test_rated_power_uprate_lifts_region3_power() -> None: + """An uprate lifts near-rated power toward the new rated, leaving deep region 2 alone.""" + rows = _rows([1000.0, 2200.0]) + out = apply_upgrades(rows, [RatedPowerChange(new_rated_power_kw=2400.0)], cp=CpCore(rated_power_kw=2300.0)) + power = out[HOT_COLUMNS.active_power].to_numpy() + assert power[0] == pytest.approx(1000.0, rel=1e-3) # deep region 2: ~unchanged + assert 2200.0 < power[1] < 2400.0 # near rated: lifted but capped at new rated + + +def test_ws_delta_scales_nacelle_wind_speed() -> None: + """An upgrade's ws_delta scales the nacelle WindSpeedMean.""" + rows = _rows([1000.0], ws=[8.0]) + out = apply_upgrades(rows, [ConstantCpChange(delta=0.0, ws_delta=0.02)], cp=CpCore()) + assert out[HOT_COLUMNS.wind_speed].iloc[0] == pytest.approx(8.0 * 1.02) + + +def test_rpm_increases_when_power_increases() -> None: + """A positive Cp change drags generator rpm up with the power increase.""" + rows = _rows([1000.0], rpm=[1392.0]) + out = apply_upgrades(rows, [ConstantCpChange(delta=0.10)], cp=CpCore()) + assert out[HOT_COLUMNS.gen_rpm].iloc[0] > 1392.0 + + +def test_upgrades_compose() -> None: + """A Cp change and a downrate combine: region-2 power rises, near-rated power clips.""" + rows = _rows([1000.0, 2200.0]) + upgrades = [ConstantCpChange(delta=0.10), RatedPowerChange(new_rated_power_kw=2000.0)] + out = apply_upgrades(rows, upgrades, cp=CpCore(rated_power_kw=2300.0)) + power = out[HOT_COLUMNS.active_power].to_numpy() + f = region2_fraction(1000.0) + assert power[0] == pytest.approx(1000.0 * (1.0 + f * 0.10)) # region 2: +Cp, below downrate + assert power[1] == pytest.approx(2000.0) # near rated: clipped to the downrate diff --git a/tests/plots/test_northing_plots.py b/tests/plots/test_northing_plots.py new file mode 100644 index 00000000..7f9d4a5c --- /dev/null +++ b/tests/plots/test_northing_plots.py @@ -0,0 +1,79 @@ +from __future__ import annotations + +from typing import TYPE_CHECKING + +import pandas as pd + +from wind_up.constants import ( + RAW_DOWNTIME_S_COL, + RAW_POWER_COL, + RAW_YAWDIR_COL, + REANALYSIS_WD_COL, +) +from wind_up.models import PlotConfig, WindUpConfig +from wind_up.northing import calc_northed_col_name +from wind_up.plots import northing_plots + +if TYPE_CHECKING: + from pathlib import Path + + +def _make_wf_df(wtg_name: str, changepoint: pd.Timestamp, northed_col: str) -> pd.DataFrame: + ts = pd.date_range(changepoint - pd.Timedelta(days=1), changepoint + pd.Timedelta(days=1), freq="10min") + return pd.DataFrame( + { + RAW_YAWDIR_COL: 100.0, + northed_col: 100.0, + REANALYSIS_WD_COL: 100.0, + RAW_POWER_COL: 500.0, + RAW_DOWNTIME_S_COL: 0.0, + }, + index=pd.MultiIndex.from_product([[wtg_name], ts], names=["TurbineName", "TimeStamp_StartFormat"]), + ) + + +class TestPlotNorthingChangepoint: + @staticmethod + def test_does_not_save_when_save_plots_false(test_homer_config: WindUpConfig, tmp_path: Path) -> None: + wtg_name = test_homer_config.asset.wtgs[0].name + changepoint = pd.Timestamp("2018-07-01 00:00:00", tz="UTC") + northed_col = calc_northed_col_name(REANALYSIS_WD_COL) + wf_df = _make_wf_df(wtg_name, changepoint, northed_col) + plots_dir = tmp_path / "plots" + + northing_plots.plot_northing_changepoint( + wf_df, + northing_turbine=wtg_name, + northed_col=northed_col, + north_ref_wd_col=REANALYSIS_WD_COL, + northing_datetime_utc=changepoint, + cfg=test_homer_config, + plot_cfg=PlotConfig(save_plots=False, show_plots=False, plots_dir=plots_dir), + ) + + assert not plots_dir.exists() + + @staticmethod + def test_saves_when_save_plots_true(test_homer_config: WindUpConfig, tmp_path: Path) -> None: + wtg_name = test_homer_config.asset.wtgs[0].name + changepoint = pd.Timestamp("2018-07-01 00:00:00", tz="UTC") + northed_col = calc_northed_col_name(REANALYSIS_WD_COL) + wf_df = _make_wf_df(wtg_name, changepoint, northed_col) + plots_dir = tmp_path / "plots" + + northing_plots.plot_northing_changepoint( + wf_df, + northing_turbine=wtg_name, + northed_col=northed_col, + north_ref_wd_col=REANALYSIS_WD_COL, + northing_datetime_utc=changepoint, + cfg=test_homer_config, + plot_cfg=PlotConfig(save_plots=True, show_plots=False, plots_dir=plots_dir), + ) + + expected = ( + plots_dir + / wtg_name + / f"{wtg_name} north_ref_wd_col={REANALYSIS_WD_COL} {changepoint.strftime('%Y-%m-%d')}.png" + ) + assert expected.is_file() diff --git a/tests/plots/test_optimize_northing_plots.py b/tests/plots/test_optimize_northing_plots.py new file mode 100644 index 00000000..d2c1307a --- /dev/null +++ b/tests/plots/test_optimize_northing_plots.py @@ -0,0 +1,79 @@ +from __future__ import annotations + +from typing import TYPE_CHECKING + +import pandas as pd +import pytest + +from wind_up.constants import RAW_POWER_COL, REANALYSIS_WD_COL, WINDFARM_YAWDIR_COL +from wind_up.models import PlotConfig, WindUpConfig +from wind_up.plots import optimize_northing_plots + +if TYPE_CHECKING: + from pathlib import Path + + +def _diff_df(north_ref_wd_col: str) -> pd.DataFrame: + ts = pd.date_range("2020-01-01", periods=20, freq="10min") + return pd.DataFrame( + { + f"yaw_diff_to_{north_ref_wd_col}": 1.0, + f"filt_diff_to_{north_ref_wd_col}": 1.0, + f"short_rolling_diff_to_{north_ref_wd_col}": 1.0, + f"long_rolling_diff_to_{north_ref_wd_col}": 1.0, + RAW_POWER_COL: 500.0, + }, + index=ts, + ) + + +class TestPlotDiffToNorthRefWd: + @staticmethod + @pytest.mark.parametrize("save_plots", [False, True]) + def test_honors_save_plots(tmp_path: Path, save_plots: bool) -> None: # noqa: FBT001 + plots_dir = tmp_path / "plots" + wtg_name = "T01" + optimize_northing_plots.plot_diff_to_north_ref_wd( + _diff_df(REANALYSIS_WD_COL), + wtg_name=wtg_name, + north_ref_wd_col=REANALYSIS_WD_COL, + loop_count=0, + plot_cfg=PlotConfig(save_plots=save_plots, show_plots=False, plots_dir=plots_dir), + ) + any_saved = plots_dir.exists() and any(plots_dir.rglob("*.png")) + assert any_saved is save_plots + + +class TestPlotYawDiffVsPower: + @staticmethod + @pytest.mark.parametrize("save_plots", [False, True]) + def test_honors_save_plots(tmp_path: Path, save_plots: bool) -> None: # noqa: FBT001 + plots_dir = tmp_path / "plots" + optimize_northing_plots.plot_yaw_diff_vs_power( + _diff_df(REANALYSIS_WD_COL), + wtg_name="T01", + north_ref_wd_col=REANALYSIS_WD_COL, + plot_cfg=PlotConfig(save_plots=save_plots, show_plots=False, plots_dir=plots_dir), + ) + any_saved = plots_dir.exists() and any(plots_dir.rglob("*.png")) + assert any_saved is save_plots + + +class TestPlotWfYawdirAndReanalysisTimeseries: + @staticmethod + @pytest.mark.parametrize("save_plots", [False, True]) + def test_honors_save_plots(test_homer_config: WindUpConfig, tmp_path: Path, save_plots: bool) -> None: # noqa: FBT001 + plots_dir = tmp_path / "plots" + wtg_name = test_homer_config.asset.wtgs[0].name + ts = pd.date_range("2020-01-01", periods=20, freq="10min") + wf_df = pd.DataFrame( + {WINDFARM_YAWDIR_COL: 100.0, REANALYSIS_WD_COL: 100.0}, + index=pd.MultiIndex.from_product([[wtg_name], ts], names=["TurbineName", "TimeStamp_StartFormat"]), + ) + optimize_northing_plots.plot_wf_yawdir_and_reanalysis_timeseries( + wf_df, + cfg=test_homer_config, + plot_cfg=PlotConfig(save_plots=save_plots, show_plots=False, plots_dir=plots_dir), + ) + any_saved = plots_dir.exists() and any(plots_dir.rglob("*.png")) + assert any_saved is save_plots diff --git a/tests/plots/test_scada_funcs_plots.py b/tests/plots/test_scada_funcs_plots.py index a228261e..b9bc6a6d 100644 --- a/tests/plots/test_scada_funcs_plots.py +++ b/tests/plots/test_scada_funcs_plots.py @@ -121,3 +121,39 @@ def test_no_reactive_power_column(caplog: pytest.LogCaptureFixture) -> None: log_msg = caplog.records[-1] assert log_msg.levelname == logging.getLevelName(logging.WARNING) assert expected_msg == log_msg.getMessage() + + +class TestPrintAndPlotCapacityFactor: + @staticmethod + def _make_scada_df(cfg) -> pd.DataFrame: # noqa: ANN001 + rows = [ + {DataColumns.turbine_name: wtg.name, DataColumns.active_power_mean: 500.0} + for wtg in cfg.asset.wtgs + for _ in range(6) + ] + return pd.DataFrame(rows) + + def test_does_not_save_when_save_plots_false(self, test_homer_config, tmp_path: Path) -> None: # noqa: ANN001 + plots_dir = tmp_path / "plots" + scada_df = self._make_scada_df(test_homer_config) + + scada_funcs_plots.print_and_plot_capacity_factor( + scada_df=scada_df, + cfg=test_homer_config, + plots_cfg=PlotConfig(save_plots=False, show_plots=False, plots_dir=plots_dir), + ) + + assert not plots_dir.exists() + + def test_saves_when_save_plots_true(self, test_homer_config, tmp_path: Path) -> None: # noqa: ANN001 + plots_dir = tmp_path / "plots" + scada_df = self._make_scada_df(test_homer_config) + + scada_funcs_plots.print_and_plot_capacity_factor( + scada_df=scada_df, + cfg=test_homer_config, + plots_cfg=PlotConfig(save_plots=True, show_plots=False, plots_dir=plots_dir), + ) + + expected = plots_dir / f"{test_homer_config.asset.name} capacity factor.png" + assert expected.is_file() diff --git a/tests/test_backend.py b/tests/test_backend.py index 00066711..e9f428cc 100644 --- a/tests/test_backend.py +++ b/tests/test_backend.py @@ -3,6 +3,7 @@ import os import subprocess import sys +from pathlib import Path _PRINT_BACKEND = "import wind_up, matplotlib; print(matplotlib.get_backend())" # import pyplot (selecting a non-interactive backend) *before* wind_up, as an interactive @@ -13,8 +14,30 @@ ) +def _pythonpath_without_sitecustomize_hooks() -> str | None: + """Return PYTHONPATH with any entry that injects a ``sitecustomize`` hook removed. + + These tests check wind_up's own backend selection in a pristine interpreter, but an IDE + can prepend a helper dir to PYTHONPATH whose ``sitecustomize.py`` runs at startup and + forces a backend, overriding MPLBACKEND. PyCharm's "Show plots in tool window" does exactly + this (``pycharm_matplotlib_backend`` calls ``matplotlib.use('module://backend_interagg')``). + Such a foreign override would leak into our subprocess and defeat what we're asserting, so + strip any PYTHONPATH dir carrying a ``sitecustomize.py``. + """ + raw = os.environ.get("PYTHONPATH") + if not raw: + return None + kept = [p for p in raw.split(os.pathsep) if p and not (Path(p) / "sitecustomize.py").is_file()] + return os.pathsep.join(kept) if kept else None + + def _backend_after(code: str, *, mplbackend: str | None = None) -> str: env = {k: v for k, v in os.environ.items() if k != "MPLBACKEND"} + sanitized_pythonpath = _pythonpath_without_sitecustomize_hooks() + if sanitized_pythonpath is None: + env.pop("PYTHONPATH", None) + else: + env["PYTHONPATH"] = sanitized_pythonpath if mplbackend is not None: env["MPLBACKEND"] = mplbackend result = subprocess.run( # noqa: S603 diff --git a/tests/test_era5.py b/tests/test_era5.py new file mode 100644 index 00000000..e89b871b --- /dev/null +++ b/tests/test_era5.py @@ -0,0 +1,114 @@ +"""Tests for wind_up.era5 (Open-Meteo ERA5 fetch + cache). + +Ported from the hill-of-towie-open-source-analysis ``test_era5_helpers.py``; adapted to +wind_up's module path and the wind_up cache-dir resolver. All offline (mocked); the real +network fetch is exercised only by callers behind a ``slow`` marker. +""" + +from pathlib import Path +from unittest.mock import MagicMock + +import numpy as np +import pandas as pd +import pytest + +from wind_up import era5 +from wind_up.era5 import _build_era5_df + + +def _make_mock_response(n_hours: int = 3) -> MagicMock: + def _make_var(val: float) -> MagicMock: + mock_var = MagicMock() + mock_var.ValuesAsNumpy.return_value = np.full(n_hours, val, dtype="float32") + return mock_var + + mock_hourly = MagicMock() + mock_hourly.Time.return_value = 1704067200 # 2024-01-01 00:00 UTC + mock_hourly.TimeEnd.return_value = 1704067200 + 3600 * n_hours + mock_hourly.Interval.return_value = 3600 + mock_hourly.Variables.side_effect = lambda i: _make_var(float(i)) + + mock_response = MagicMock() + mock_response.Hourly.return_value = mock_hourly + return mock_response + + +class TestBuildEra5Df: + def test_columns_match_fields(self) -> None: + fields = ["wind_speed_10m", "wind_direction_10m"] + df = _build_era5_df(_make_mock_response(), fields) + assert list(df.columns) == ["wind_speed_10m", "wind_direction_10m"] + + def test_row_count_matches_hours(self) -> None: + fields = ["wind_speed_10m"] + df = _build_era5_df(_make_mock_response(n_hours=5), fields) + assert len(df) == 5 + + def test_index_is_utc_datetimeindex(self) -> None: + fields = ["wind_speed_10m"] + df = _build_era5_df(_make_mock_response(), fields) + assert df.index.dtype == "datetime64[ns, UTC]" + + def test_timestamp_starts_at_expected_value(self) -> None: + fields = ["wind_speed_10m"] + df = _build_era5_df(_make_mock_response(), fields) + assert df.index[0] == pd.Timestamp("2024-01-01", tz="UTC") + + def test_field_values_are_propagated(self) -> None: + fields = ["wind_speed_10m", "wind_direction_10m"] + df = _build_era5_df(_make_mock_response(), fields) + assert df["wind_speed_10m"].iloc[0] == pytest.approx(0.0) # index 0 + assert df["wind_direction_10m"].iloc[0] == pytest.approx(1.0) # index 1 + + +class TestGetEra5HourlyDf: + def test_end_date_defaults_to_today(self, monkeypatch: pytest.MonkeyPatch) -> None: + captured: dict[str, object] = {} + + def fake_cache_path(*args: object, **_kwargs: object) -> MagicMock: + # positional signature mirrors _era5_cache_path(lat, lon, start_date, end_date, fields) + captured["end_date"] = args[3] + mock_path = MagicMock() + mock_path.exists.return_value = True + return mock_path + + monkeypatch.setattr(era5, "_era5_cache_path", fake_cache_path) + monkeypatch.setattr(era5.pd, "read_parquet", lambda _p: pd.DataFrame()) + era5.get_era5_hourly_df(lat=1.0, lon=2.0) + today = pd.Timestamp.now(tz="UTC").normalize().strftime("%Y-%m-%d") + assert captured["end_date"] == today + + def test_reads_from_cache_when_present(self, monkeypatch: pytest.MonkeyPatch) -> None: + cached = pd.DataFrame({"wind_speed_100m": [1.0, 2.0]}) + + mock_path = MagicMock() + mock_path.exists.return_value = True + monkeypatch.setattr(era5, "_era5_cache_path", lambda *_a, **_k: mock_path) + monkeypatch.setattr(era5.pd, "read_parquet", lambda _p: cached) + + out = era5.get_era5_hourly_df(lat=1.0, lon=2.0, start_date="2020-01-01", end_date="2020-01-02") + pd.testing.assert_frame_equal(out, cached) + + +class TestCacheDir: + def test_env_override(self, monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: + monkeypatch.setenv("WIND_UP_CACHE_DIR", str(tmp_path)) + assert era5._resolve_cache_dir(None) == tmp_path # noqa: SLF001 + + def test_explicit_arg_wins(self, monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: + monkeypatch.setenv("WIND_UP_CACHE_DIR", str(tmp_path / "env")) + explicit = tmp_path / "explicit" + assert era5._resolve_cache_dir(explicit) == explicit # noqa: SLF001 + + def test_cache_path_is_deterministic(self, tmp_path: Path) -> None: + args = { + "lat": 57.5, + "lon": -3.25, + "start_date": "2020-01-01", + "end_date": "2020-02-01", + "fields": ["wind_speed_100m", "wind_direction_100m"], + } + p1 = era5._era5_cache_path(**args, cache_dir=tmp_path) # noqa: SLF001 + p2 = era5._era5_cache_path(**args, cache_dir=tmp_path) # noqa: SLF001 + assert p1 == p2 + assert p1.suffix == ".parquet" diff --git a/tests/test_main_analysis.py b/tests/test_main_analysis.py index 21f96aab..762f4ec2 100644 --- a/tests/test_main_analysis.py +++ b/tests/test_main_analysis.py @@ -1,7 +1,9 @@ +import logging import math import numpy as np import pandas as pd +import pytest from pandas.testing import assert_frame_equal from wind_up.constants import TIMESTAMP_COL @@ -49,6 +51,38 @@ def test_toggle_pairing_filter_method_none() -> None: assert_frame_equal(filt_post_df, post_df) +def test_toggle_pairing_filter_method_none_reports_zero_removed(caplog: pytest.LogCaptureFixture) -> None: + # 'none' is a no-op, so it must report 0 rows removed even when the inputs contain NaN rows in the + # required columns. before/after are both measured on the dropna basis, so the count can never go + # negative (regression: it previously logged "removed -N [-M%]" because 'after' counted the raw + # frame incl. NaN rows while 'before' counted only the valid rows). + tstamps = pd.date_range(start="2021-01-01 00:00:00", tz="UTC", periods=9, freq="10min") + detrend_ws_col, test_pw_col, ref_wd_col = "ref_ws_detrended", "test_pw_clipped", "ref_YawAngleMean" + df = pd.DataFrame( + data={ + detrend_ws_col: [5.1, 5.1, 5.1, 0.0, 0.0, 0.0, np.nan, 5.1, 5.1], + test_pw_col: [5.1, 5.1, np.nan, 0.0, 0.0, 0.0, 5.1, 5.1, 5.1], + ref_wd_col: [5.1, 5.1, 5.1, 0.0, 0.0, 0.0, 5.1, 5.1, np.nan], # 3 distinct rows NaN in required cols + }, + index=tstamps, + ) + with caplog.at_level(logging.INFO, logger="wind_up.main_analysis"): + _toggle_pairing_filter( + pre_df=df, + post_df=df, + pairing_filter_method="none", + pairing_filter_timedelta_seconds=0, + detrend_ws_col=detrend_ws_col, + test_pw_col=test_pw_col, + ref_wd_col=ref_wd_col, + timebase_s=600, + ) + removed_msgs = [r.message for r in caplog.records if "pairing filter" in r.message] + assert len(removed_msgs) == 2 # one for pre_df, one for post_df + assert all("removed 0 " in m for m in removed_msgs), removed_msgs + assert not any("removed -" in m for m in removed_msgs), removed_msgs + + def test_toggle_pairing_filter_method_any_within_timedelta() -> None: pre_tstamps = pd.date_range(start="2021-01-01 00:00:00", tz="UTC", periods=9, freq="10min") post_tstamps = pd.date_range(start=pre_tstamps.max() + pd.Timedelta("10min"), tz="UTC", periods=9, freq="10min") diff --git a/uv.lock b/uv.lock index e61cb671..4fc76270 100644 --- a/uv.lock +++ b/uv.lock @@ -162,6 +162,20 @@ css = [ { name = "tinycss2" }, ] +[[package]] +name = "cattrs" +version = "26.1.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "attrs" }, + { name = "exceptiongroup", marker = "python_full_version < '3.11'" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/a0/ec/ba18945e7d6e55a58364d9fb2e46049c1c2998b3d805f19b703f14e81057/cattrs-26.1.0.tar.gz", hash = "sha256:fa239e0f0ec0715ba34852ce813986dfed1e12117e209b816ab87401271cdd40", size = 495672, upload-time = "2026-02-18T22:15:19.406Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/80/56/60547f7801b97c67e97491dc3d9ade9fbccbd0325058fd3dfcb2f5d98d90/cattrs-26.1.0-py3-none-any.whl", hash = "sha256:d1e0804c42639494d469d08d4f26d6b9de9b8ab26b446db7b5f8c2e97f7c3096", size = 73054, upload-time = "2026-02-18T22:15:17.958Z" }, +] + [[package]] name = "certifi" version = "2026.5.20" @@ -710,6 +724,15 @@ automl = [ { name = "xgboost" }, ] +[[package]] +name = "flatbuffers" +version = "25.9.23" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/9d/1f/3ee70b0a55137442038f2a33469cc5fddd7e0ad2abf83d7497c18a2b6923/flatbuffers-25.9.23.tar.gz", hash = "sha256:676f9fa62750bb50cf531b42a0a2a118ad8f7f797a511eda12881c016f093b12", size = 22067, upload-time = "2025-09-24T05:25:30.106Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/ee/1b/00a78aa2e8fbd63f9af08c9c19e6deb3d5d66b4dda677a0f61654680ee89/flatbuffers-25.9.23-py2.py3-none-any.whl", hash = "sha256:255538574d6cb6d0a79a17ec8bc0d30985913b87513a01cce8bcdb6b4c44d0e2", size = 30869, upload-time = "2025-09-24T05:25:28.912Z" }, +] + [[package]] name = "fonttools" version = "4.63.0" @@ -954,6 +977,69 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/9a/93/242e2eab5fe682ffcb8b0084bde703a41d51e17ee0f3a31ff0d9d813620a/jedi-0.20.0-py2.py3-none-any.whl", hash = "sha256:7bdd9c2634f56713299976f4cbd59cb3fa92165cc5e05ea811fb253480728b67", size = 4884812, upload-time = "2026-05-01T23:38:43.919Z" }, ] +[[package]] +name = "jh2" +version = "5.0.13" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/47/b1/b2b7389b2e0ddac90a1aecbf4a761db8790de85dace7695c01173ed083cc/jh2-5.0.13.tar.gz", hash = "sha256:f8c78cffb3a35c4410513c3eb7989de36028c84277c04f07c97909dd94c23a75", size = 7322505, upload-time = "2026-05-29T05:21:43.516Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/c4/96/c38d3555cc9bbb1c309c896e99dbb0b1f76308bed7f2d6226cdbeaeb3eee/jh2-5.0.13-cp313-cp313t-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:2195557f5953ceee743d0eaab0b09ec537bbc1b9a2c6c1463106bfca9b03805f", size = 601454, upload-time = "2026-05-29T05:19:23.763Z" }, + { url = "https://files.pythonhosted.org/packages/4e/84/a3043ddaf636cec0f4c3aa179a9ef3a59db3d06f770ab6007133ca31bc5f/jh2-5.0.13-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:4e734f8a5cf002316cc81b80d8a91a0f7c62d631270abd913746644fb9ae288d", size = 384187, upload-time = "2026-05-29T05:19:25.504Z" }, + { url = "https://files.pythonhosted.org/packages/fb/8b/72dba3fb64bf01355a0bbd78cbe0a2635759d4a6f273c1b118760ca8ecf0/jh2-5.0.13-cp313-cp313t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:ba5e545b3209e8aa5b6af88e7032814473e57d60d32d3fc6b6185453733a4b26", size = 389346, upload-time = "2026-05-29T05:19:26.926Z" }, + { url = "https://files.pythonhosted.org/packages/8e/2e/6f07e42de6101d4555f14be1e4f42a056c9a44d7ee8f745b3c188c96cbe0/jh2-5.0.13-cp313-cp313t-manylinux_2_17_ppc64.manylinux2014_ppc64.whl", hash = "sha256:2a37164eb8f2462268e6252db93b009cb685fa060b62ad217bf1b333b450c68d", size = 510278, upload-time = "2026-05-29T05:19:28.175Z" }, + { url = "https://files.pythonhosted.org/packages/c6/a5/85682dcc959682aede0f80206a16574c51af51ed26eb1e87252fd31c6ded/jh2-5.0.13-cp313-cp313t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:b4eae2d5bc1a91033450fd4381c04f33db7d4c9cfd31046c4708c8c1323c06d0", size = 503090, upload-time = "2026-05-29T05:19:29.341Z" }, + { url = "https://files.pythonhosted.org/packages/c2/4c/a8a301d494c0a1c2859951d44b4fa54f3b2db91826c719ce5be8b662c5c8/jh2-5.0.13-cp313-cp313t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:40d0a10036cdf8bf1a36b07c10b69ad0d08ce31d99b8962e3d055ee417fe3423", size = 402843, upload-time = "2026-05-29T05:19:30.505Z" }, + { url = "https://files.pythonhosted.org/packages/21/5f/5f9e0933f458a89005af3b2aa7d1c61cc3cc57c8a633fd54a95acff85121/jh2-5.0.13-cp313-cp313t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:0ad4d68340e38a8474a48c424e2929d3e2e82e35e4e8ff1188decb1985d1b4d7", size = 386327, upload-time = "2026-05-29T05:19:31.988Z" }, + { url = "https://files.pythonhosted.org/packages/75/3d/353ec4a6763e8b6be8d162e6b23f7e1499e75b1bb934cee1e5b5fc062f1e/jh2-5.0.13-cp313-cp313t-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:3f8c5e81254b986f8abe9fd974e7ed595aeaead3f3066ce9ed95dfff236ceca9", size = 406978, upload-time = "2026-05-29T05:19:33.438Z" }, + { url = "https://files.pythonhosted.org/packages/4c/e5/cca58b58cb8bab1891f45da43ef91b498ff93d573a1a2836af11962ccfb8/jh2-5.0.13-cp313-cp313t-musllinux_1_1_aarch64.whl", hash = "sha256:da030729eb8ef386569e78c10e2429adb10026ce82989fdeb906b037314d66c7", size = 560030, upload-time = "2026-05-29T05:19:34.879Z" }, + { url = "https://files.pythonhosted.org/packages/c7/c9/c3391d8607781016daa1881d8d80f81e0cf0302ff06c86342c492279cf48/jh2-5.0.13-cp313-cp313t-musllinux_1_1_armv7l.whl", hash = "sha256:a5d2ceb2e9d42ac236f0da7b2f9e8237818d32a10d892b06f30f7160ee4365ae", size = 664407, upload-time = "2026-05-29T05:19:36.138Z" }, + { url = "https://files.pythonhosted.org/packages/e3/b4/3375744e2f33d097da241dfd5262c9997a87c2b948db2385fbf100c7dd12/jh2-5.0.13-cp313-cp313t-musllinux_1_1_i686.whl", hash = "sha256:1f4dffe1c6ed0d4b0d2d9a45d064622bb786cbf87db35eac469dd4c376ac6fb6", size = 625242, upload-time = "2026-05-29T05:19:37.93Z" }, + { url = "https://files.pythonhosted.org/packages/9e/2f/e33763b1f66817bd8c848777f05cf3781ad9ee1c642741458bfb66da1921/jh2-5.0.13-cp313-cp313t-musllinux_1_1_x86_64.whl", hash = "sha256:bdeb48657db508e8bc299744aa431798efdf090feb3959cf0123167c9251ebb0", size = 591124, upload-time = "2026-05-29T05:19:39.373Z" }, + { url = "https://files.pythonhosted.org/packages/90/92/5a8de2f6936c7af77555299732795f4607d2cefb974ad485acf25b3f998b/jh2-5.0.13-cp313-cp313t-win32.whl", hash = "sha256:b41867d5decd7cd62e12cdacfc45c5d8f6eae872578ae01d78900174d5e47b82", size = 238101, upload-time = "2026-05-29T05:19:40.917Z" }, + { url = "https://files.pythonhosted.org/packages/e7/71/0f7ba74bb6681e60e871284772a216a50cfd63f4f62ef52edd13d8a7c057/jh2-5.0.13-cp313-cp313t-win_amd64.whl", hash = "sha256:62aee4bd128e17871980e4847cc4b94b67651de612bc96dd3d7ebf647c68a6e4", size = 245923, upload-time = "2026-05-29T05:19:42.338Z" }, + { url = "https://files.pythonhosted.org/packages/b1/dc/11434a6deee70630f5f38cc40813692e1faadd4259f242f4f810f35709eb/jh2-5.0.13-cp313-cp313t-win_arm64.whl", hash = "sha256:267aaa7a96f54452e1e07c7701799bc7817e1f3bb2de09301b2c492841e4c98e", size = 242596, upload-time = "2026-05-29T05:19:43.853Z" }, + { url = "https://files.pythonhosted.org/packages/75/f3/8c151f371f7962902e6f1f494b50abc1b8be27305a9d12cb88cf049c5a80/jh2-5.0.13-cp37-abi3-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:3f54095cde010ff4e52f23dda92b31d3d3160c64f5f23fc6e181f8c7938afd33", size = 618581, upload-time = "2026-05-29T05:20:05.043Z" }, + { url = "https://files.pythonhosted.org/packages/71/3f/a2d5458586c024e0076ae8182dcbb71450bd9d4dd65ffdaa659313461bd4/jh2-5.0.13-cp37-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:f810dd02b4a9647eacbe4eab617111c0e3ffc68e923e3ff172948db601d2ba1e", size = 392776, upload-time = "2026-05-29T05:20:06.178Z" }, + { url = "https://files.pythonhosted.org/packages/7e/1f/cf205104cfcef4bd26aa2b736f36ddf3739b69ddaf14d23a5f7697673d74/jh2-5.0.13-cp37-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:29c23f3937f257beb5cd5b17f8727404f1267ff25d26e54c0a35d5defd0c895e", size = 397943, upload-time = "2026-05-29T05:20:07.479Z" }, + { url = "https://files.pythonhosted.org/packages/cf/82/bd218132e5a081293cf3d7de5cf41d871f3f88808c23f6c6aebbe073fb5e/jh2-5.0.13-cp37-abi3-manylinux_2_17_ppc64.manylinux2014_ppc64.whl", hash = "sha256:c697b250a935030a128b45b41f507a4d33a082d8b8e12311c7e73fbb3970f057", size = 520351, upload-time = "2026-05-29T05:20:08.936Z" }, + { url = "https://files.pythonhosted.org/packages/67/7c/78b564899237909ddf4103569be59ea31b6c00f8b4f1c178cc57c6e4e45d/jh2-5.0.13-cp37-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:47d3203ed6fec39df4922764817979fad3a13d22cf750adc9cf9dc2b3399eacf", size = 511162, upload-time = "2026-05-29T05:20:10.239Z" }, + { url = "https://files.pythonhosted.org/packages/db/3a/b59157363139436cb60620353aa8b687197c9a4ec1eb87952004a8843e6e/jh2-5.0.13-cp37-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:a5146468a76bbe8f4035f23a78813c5c433c70c0ca44eb0ba5558627a9e644b8", size = 411838, upload-time = "2026-05-29T05:20:11.515Z" }, + { url = "https://files.pythonhosted.org/packages/15/a4/18b35000b00d9fb72003e0eacf66b3af828a0bb897c21a287a33d2025138/jh2-5.0.13-cp37-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:85d3d338284fbfbe173243c3cc62e510c6fa0d6d023e67da2292bd7578c734e3", size = 395975, upload-time = "2026-05-29T05:20:13.157Z" }, + { url = "https://files.pythonhosted.org/packages/ab/42/f017b6b7b666e6ef615f7c64fd748201d0b20edeed094182c4f37f23a6e8/jh2-5.0.13-cp37-abi3-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:641f65f5414b8990251bd7609513a696831558f69d2ca77d65dc3f7087f31267", size = 417493, upload-time = "2026-05-29T05:20:14.394Z" }, + { url = "https://files.pythonhosted.org/packages/bb/fc/054443bd40a9827ade99508482185692530cf56a50d4ad8ed2aba761f11d/jh2-5.0.13-cp37-abi3-musllinux_1_1_aarch64.whl", hash = "sha256:cf3dc6f7c14340241260bf63e5d6006b2ee7db45dafcb39c16fdf164ea0240a0", size = 568373, upload-time = "2026-05-29T05:20:15.914Z" }, + { url = "https://files.pythonhosted.org/packages/47/48/dde093626bb976722c940053ceaa82407bd358e07fa6a1690590456d21b5/jh2-5.0.13-cp37-abi3-musllinux_1_1_armv7l.whl", hash = "sha256:fb8e37ecf37382c02af83da521df5a77561441bc4cbb93b00627e015b7bbf349", size = 673766, upload-time = "2026-05-29T05:20:17.136Z" }, + { url = "https://files.pythonhosted.org/packages/96/94/edae999301bb28738e17cf54ecd50b7d7d9fd96d9536376bbeb9b8498526/jh2-5.0.13-cp37-abi3-musllinux_1_1_i686.whl", hash = "sha256:296ee44dfad08ccc3857daf433dfa4a555d5b45f2df39503ccdd90ff10b32943", size = 635045, upload-time = "2026-05-29T05:20:18.633Z" }, + { url = "https://files.pythonhosted.org/packages/c3/5b/d9fd0fc71e43c37c30d581affbb7371fa8177ce211b663457939bbed9e61/jh2-5.0.13-cp37-abi3-musllinux_1_1_x86_64.whl", hash = "sha256:1a75eed545420a31ea356e8046b8d86fcf694234dd1035191734a03720fb8a44", size = 599870, upload-time = "2026-05-29T05:20:20.117Z" }, + { url = "https://files.pythonhosted.org/packages/b1/ec/66c17d8aec77d41c8f93145dd7faa7138b9f1bc370fa685008bf92d75f9f/jh2-5.0.13-cp37-abi3-win32.whl", hash = "sha256:4812710810010c428db04e9290a13117f7b7c5a3a663b46a0586fe101cf25c9d", size = 245231, upload-time = "2026-05-29T05:20:21.275Z" }, + { url = "https://files.pythonhosted.org/packages/5b/3c/107f16a84c8becef5281040fdb50681f06e4615bb4a754f211abe7d61a6f/jh2-5.0.13-cp37-abi3-win_amd64.whl", hash = "sha256:5d61bc0a62b4eee11039bf29d4bdbb26efcff760dfb9d8e0739740028aa52b0f", size = 252085, upload-time = "2026-05-29T05:20:22.426Z" }, + { url = "https://files.pythonhosted.org/packages/47/7b/e8baeef7655449a6c9abee0165fcb1c63d12ec5432b8d810ff5910184690/jh2-5.0.13-cp37-abi3-win_arm64.whl", hash = "sha256:3f74dad9036e599eed51576492b00955237b86cf886f96844708a81cd0f5c710", size = 249139, upload-time = "2026-05-29T05:20:23.558Z" }, + { url = "https://files.pythonhosted.org/packages/64/5c/5f006d5bec874c9b12f36f20fbdd3fce545994b3d43eb216266d36c1cd09/jh2-5.0.13-pp310-pypy310_pp73-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:72122b850d58c4ae0612c9b9bbabeddbb092b503eb39d87249da323ba1b074a8", size = 614502, upload-time = "2026-05-29T05:20:24.757Z" }, + { url = "https://files.pythonhosted.org/packages/61/9e/ee99e108c6b458ec3298a76b59385ffe0e8f82d93ad30efbfb23632f6991/jh2-5.0.13-pp310-pypy310_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:d5c4d4ef923464dc5b39d39b1a41209a23e387698a254e81813d0757f76ff00a", size = 390823, upload-time = "2026-05-29T05:20:25.959Z" }, + { url = "https://files.pythonhosted.org/packages/3d/64/740cc33301c84ba79304567c3d199bf269805b7baebe8e0082739b97524a/jh2-5.0.13-pp310-pypy310_pp73-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:bc238641d3bc35dc79965f330c7f0f16ca992cf999c8892dce0bd36c32056eee", size = 396388, upload-time = "2026-05-29T05:20:27.202Z" }, + { url = "https://files.pythonhosted.org/packages/e9/ed/522f67aa1946c27ef66dd85408fb16b94fe280b84833f06455e2ead119bc/jh2-5.0.13-pp310-pypy310_pp73-manylinux_2_17_ppc64.manylinux2014_ppc64.whl", hash = "sha256:5f9a82f90a21660fd0313a81c838bff89a13c7921c317abfc3117e49f9d46c84", size = 517651, upload-time = "2026-05-29T05:20:28.353Z" }, + { url = "https://files.pythonhosted.org/packages/ea/5b/e5d8568a60b265e6b34672269407ff6d3a6c43be821b433be54dc8ce63cd/jh2-5.0.13-pp310-pypy310_pp73-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:a1ef1d6c90389c366ade34bc6b100e823569c60bbb0b0f2cc3b308031090b252", size = 508661, upload-time = "2026-05-29T05:20:29.739Z" }, + { url = "https://files.pythonhosted.org/packages/1b/ea/1490d995d08cfb67350625f306297b9407ce7012ddbb83f97b1d18c63be8/jh2-5.0.13-pp310-pypy310_pp73-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:95d72124353a530ffa5a65f1f94a4396521f5472f0fde79b93dd66a3addaadff", size = 409884, upload-time = "2026-05-29T05:20:31.016Z" }, + { url = "https://files.pythonhosted.org/packages/c0/b1/f8a48046f17d927d5847504b1199212755baba5f1d842f9a0bc7abd5d7db/jh2-5.0.13-pp310-pypy310_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:50d224b482e1e18fa6bbe90d05125993122c5530b0b83f48829d7a535e2b60f2", size = 393947, upload-time = "2026-05-29T05:20:32.222Z" }, + { url = "https://files.pythonhosted.org/packages/12/f0/001ea64504515c7990394ebaedb285611c7dd8786ad2005a4220e4e0575e/jh2-5.0.13-pp310-pypy310_pp73-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:49531e0042b3a4d35d34ddb892bb734d77969c9b787e6f2f0253cb10b498d4e8", size = 414876, upload-time = "2026-05-29T05:20:33.519Z" }, + { url = "https://files.pythonhosted.org/packages/1c/4d/25126b7a084c946326d3fa73d24473853229dd8d3f5f0e26d3ef6ded36f5/jh2-5.0.13-pp310-pypy310_pp73-musllinux_1_1_aarch64.whl", hash = "sha256:a8bfbc5f77e95f485bf046d4a19bb773866cdc9d1b21c27a194366a55d7fe361", size = 566960, upload-time = "2026-05-29T05:20:35.12Z" }, + { url = "https://files.pythonhosted.org/packages/bf/bf/2bd0a1eb5f8cb1e516ad774efa827647ade125aa9bf658944ee5366dbb60/jh2-5.0.13-pp310-pypy310_pp73-musllinux_1_1_armv7l.whl", hash = "sha256:3e5e73bbfcf0690561a1e2bf8b2d8f1e0bfd1aca67ebbed2a753cf9440f912ef", size = 671979, upload-time = "2026-05-29T05:20:36.383Z" }, + { url = "https://files.pythonhosted.org/packages/4c/bb/e58646c5764ff6f6c1ee9f437c43247a3043cc7cb524d9777368c390ba32/jh2-5.0.13-pp310-pypy310_pp73-musllinux_1_1_i686.whl", hash = "sha256:4efed5de773df44042a07a50ea8dc476a4c120bfd3fcc4fd7a7ea145196832b2", size = 633173, upload-time = "2026-05-29T05:20:37.632Z" }, + { url = "https://files.pythonhosted.org/packages/3e/54/0bcf70845cc7ae1c0cf0da56f81460bbc2f774a03708d0fbfa164439d684/jh2-5.0.13-pp310-pypy310_pp73-musllinux_1_1_x86_64.whl", hash = "sha256:a915e77adc126bd79867b7cfecc930021ef1a2599475dda058ae80537a7068ec", size = 597680, upload-time = "2026-05-29T05:20:38.919Z" }, + { url = "https://files.pythonhosted.org/packages/5b/b2/7f5e5305e261df12244d28eb5d8150affbe6e93e3d301df7d71ca872d89a/jh2-5.0.13-pp311-pypy311_pp73-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:f3768fa39d0c5a2578b285e49e8d42230bb182bb2f7984840549f89c04ce4a56", size = 608031, upload-time = "2026-05-29T05:20:40.212Z" }, + { url = "https://files.pythonhosted.org/packages/ac/8a/68b232de0eecc6e52548d9ef7ca20baf18d8cfb5cc64186d8015dda0188e/jh2-5.0.13-pp311-pypy311_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:3ff9abf4b4769c9c8117d929f87c90415cb65f54faa22f64fbf044427127e65f", size = 388655, upload-time = "2026-05-29T05:20:41.707Z" }, + { url = "https://files.pythonhosted.org/packages/2c/31/dacd2315290e309df81b931434f30e3e73e0fededb9e0487667a5287635f/jh2-5.0.13-pp311-pypy311_pp73-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:797e8846a4c7a18a9b3abdd437ca2fa06d39223df7a6fdcae57adeb7cc3d7a95", size = 392725, upload-time = "2026-05-29T05:20:42.912Z" }, + { url = "https://files.pythonhosted.org/packages/04/3e/a9fa9e093e984fd778bedcab2246f16b0e5a5b147466dd1d5784348239b5/jh2-5.0.13-pp311-pypy311_pp73-manylinux_2_17_ppc64.manylinux2014_ppc64.whl", hash = "sha256:31fec2270a204669fd5183465e3ef6a666a0539af61bf8e995df91f16c945c0b", size = 515975, upload-time = "2026-05-29T05:20:44.213Z" }, + { url = "https://files.pythonhosted.org/packages/48/ab/2bd00b443b4cb018ed522b1ecadb1edcfbda2226b5f1dc4b733488cd1a9f/jh2-5.0.13-pp311-pypy311_pp73-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:f48eeba6c0bb91321dbefc60527d945f20ccee3f0066bb3ba0d9b71120ec9417", size = 505682, upload-time = "2026-05-29T05:20:45.38Z" }, + { url = "https://files.pythonhosted.org/packages/83/bd/63d467fda5a86abec281ebfbe7d21f49689d97e0473800f11815fe92dd10/jh2-5.0.13-pp311-pypy311_pp73-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:05849bda038c2f8cee50559e3fd363af39c80fb1f24ecf947ccbc7e1a890683f", size = 406385, upload-time = "2026-05-29T05:20:46.568Z" }, + { url = "https://files.pythonhosted.org/packages/71/8e/146902bf3b16bcca18daf217ee9a1af34a47f6913f28c24bb403faf9832d/jh2-5.0.13-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:882b13542d7bd20f79cb92fc31ecc9587037278d0a6b36c14b734f9ac2cb8890", size = 391626, upload-time = "2026-05-29T05:20:47.916Z" }, + { url = "https://files.pythonhosted.org/packages/4b/bc/d500231f338aa889b29cdab977d99bbb94bec198f7fa0adc741b51118446/jh2-5.0.13-pp311-pypy311_pp73-manylinux_2_5_i686.manylinux1_i686.whl", hash = "sha256:84d42f63b1e7368194704159637e534280540369b35758a4daa712bf277e80ef", size = 411317, upload-time = "2026-05-29T05:20:49.336Z" }, + { url = "https://files.pythonhosted.org/packages/08/b8/0f7e84ab491aee9348731e4ef266595b8153aa9da63d5376e96970fb3795/jh2-5.0.13-pp311-pypy311_pp73-musllinux_1_1_aarch64.whl", hash = "sha256:f6e95bfee4a72a7eb6fb4ca057ac5da297df68d302c63c300a5f400a0003fe31", size = 564695, upload-time = "2026-05-29T05:20:50.604Z" }, + { url = "https://files.pythonhosted.org/packages/8e/fc/776a56566dacc3918f51bbbaebab6c23a2e979a70b40ba1f526272f7b9c1/jh2-5.0.13-pp311-pypy311_pp73-musllinux_1_1_armv7l.whl", hash = "sha256:6eaa9919f7faf162d2acf66e1632c4588840589eb0030688907c9eb6a5ecae17", size = 668164, upload-time = "2026-05-29T05:20:51.905Z" }, + { url = "https://files.pythonhosted.org/packages/9c/54/ebcc2970d6ba34fdc6be0b8fac07a6fefea7e07e704654fd6dcc33d57f53/jh2-5.0.13-pp311-pypy311_pp73-musllinux_1_1_i686.whl", hash = "sha256:7586a835ed5e213593e35e0d6f90197dd76412fdcba62e05b15df72806af1e2b", size = 629234, upload-time = "2026-05-29T05:20:53.281Z" }, + { url = "https://files.pythonhosted.org/packages/cc/a7/cfa24ecc4f3a70d16d4a33557b57f1c6673fd9fa0fd10b7a932342bb27d4/jh2-5.0.13-pp311-pypy311_pp73-musllinux_1_1_x86_64.whl", hash = "sha256:6d84d87f6c1b2131273b8078f5863d980269fa53139d7f7fa76e6099e2a00bd1", size = 594400, upload-time = "2026-05-29T05:20:54.63Z" }, + { url = "https://files.pythonhosted.org/packages/ba/1e/be44f6e9ff5689fcdf9a2c0ed30880fd5da303676d89d6457a603450b7b1/jh2-5.0.13-py3-none-any.whl", hash = "sha256:2b311414989da48a155721fe47d0dc3e3b75360efd62d779aaff72d786be91cf", size = 98909, upload-time = "2026-05-29T05:21:42.086Z" }, +] + [[package]] name = "jinja2" version = "3.1.6" @@ -1590,6 +1676,20 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/c5/3c/3179b85b0e1c3659f0369940200cd6d0fa900e6cefcc7ea0bc6dd0e29ffb/nest_asyncio2-1.7.2-py3-none-any.whl", hash = "sha256:f5dfa702f3f81f6a03857e9a19e2ba578c0946a4ad417b4c50a24d7ba641fe01", size = 7843, upload-time = "2026-02-13T00:34:02.691Z" }, ] +[[package]] +name = "niquests" +version = "3.19.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "charset-normalizer" }, + { name = "urllib3-future" }, + { name = "wassima", marker = "sys_platform != 'emscripten'" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/c5/6a/b3b4d63a2d80a067dd87d6e01e071733df128b95d6806d2eb7852c35e4d2/niquests-3.19.1.tar.gz", hash = "sha256:2c34591744c7ade45f5f3a65a637cf1366d399eb514200005a8823de0d66b91e", size = 1035701, upload-time = "2026-06-08T07:53:15.754Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/80/27/498b676bf5e4824d84c6bae37bf948c164bcad0484b30851a935f165756b/niquests-3.19.1-py3-none-any.whl", hash = "sha256:ed04a8e2813f0b12f2ec82982c500fd115499b88d1abfa689fc9116ecdc68ede", size = 211349, upload-time = "2026-06-08T07:53:13.938Z" }, +] + [[package]] name = "notebook" version = "7.5.7" @@ -1754,6 +1854,31 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/50/32/e7ffa9c324ae260e5dbb4af2cd557bf7a8d155c8ac7b79a785fe1796fb92/nvidia_nccl_cu12-2.30.7-py3-none-manylinux_2_18_x86_64.whl", hash = "sha256:8ce1b8213f61f2bfac132e6df890af6450b77cbd140c6ce4e98cb0c2d8e678c9", size = 303361239, upload-time = "2026-06-09T03:24:53.816Z" }, ] +[[package]] +name = "openmeteo-requests" +version = "1.7.5" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "niquests" }, + { name = "openmeteo-sdk" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/f0/dd/b645df0f975c0bca1415fc7d2920615f35d4f8f092e9f6012c234ce212f5/openmeteo_requests-1.7.5.tar.gz", hash = "sha256:5557b5df957aa00f5e673d85ddc47f8b5bcadde6c72a98c0ec1999b538db39f9", size = 5752353, upload-time = "2026-01-19T16:42:20.006Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/f9/ac/5fcc90394486247405ea5f12623de4bd93b4ab31a34e6b782ad1a9834d8c/openmeteo_requests-1.7.5-py3-none-any.whl", hash = "sha256:790cfd7942b030901696c9b0e84cebf21b6593b95fe640d0ab223ea1b6328190", size = 7149, upload-time = "2026-01-19T16:42:18.241Z" }, +] + +[[package]] +name = "openmeteo-sdk" +version = "1.27.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "flatbuffers" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/61/11/68390b5ce2124d07b5aa55ad9ed9839146016856f319d31cbe3ca94ac84f/openmeteo_sdk-1.27.2.tar.gz", hash = "sha256:158444c01572a354c7a6bf1e6460040d9ed19ce44ce5ca43cf8679af2ec4148c", size = 12519, upload-time = "2026-06-15T16:20:24.26Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/74/04/8bed856a877f6e95fec9e2cba4195b4712ac03dff43de43631d4e87af75c/openmeteo_sdk-1.27.2-py3-none-any.whl", hash = "sha256:1dad4b26b37bcb76430e279772431865cb21e75d6f52330f58b1d7508a15bab3", size = 19591, upload-time = "2026-06-15T16:20:23.099Z" }, +] + [[package]] name = "overrides" version = "7.7.0" @@ -2404,6 +2529,74 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/01/1b/5dbe84eefc86f48473947e2f41711aded97eecef1231f4558f1f02713c12/pyzmq-27.1.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:c9f7f6e13dff2e44a6afeaf2cf54cee5929ad64afaf4d40b50f93c58fc687355", size = 544862, upload-time = "2025-09-08T23:09:56.509Z" }, ] +[[package]] +name = "qh3" +version = "1.9.2" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/63/4c/caae9fe409e81ebd495e9b2bf1b3121e8bb644898a5e30248acb7e9838cf/qh3-1.9.2.tar.gz", hash = "sha256:c6c92f63c2ec292256b5a5ed9345c42344bdaca2e55ec795623987a563aea19c", size = 344428, upload-time = "2026-06-05T06:42:46.055Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1b/9f/757ad02a8fc67c6ab3faa94568917307682382dcad15778da5b56dae5fef/qh3-1.9.2-cp313-cp313t-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:a39bf34814f317863d0e84750faf5533d699bd88387d4b2882d2f15a1f61f567", size = 4421315, upload-time = "2026-06-05T06:38:41.436Z" }, + { url = "https://files.pythonhosted.org/packages/2c/25/87b7e25af993bfafec1b426bffb27bc648f8e9398ab372bf4e8bd74078e2/qh3-1.9.2-cp313-cp313t-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:408605146ffc03247bef22ff02fcca5115f01e995921760a23264b1077fbde94", size = 2160859, upload-time = "2026-06-05T06:38:44.215Z" }, + { url = "https://files.pythonhosted.org/packages/14/1e/1eaaf4388c0efc416f444bd7fa9f2be86315a1a2ca6b1a74db4154fbd1d4/qh3-1.9.2-cp313-cp313t-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:5d87f2ade2e60f6d7dc6deb6600feacf92b29c838ee3dd2d0e03a4b34f631d83", size = 1857395, upload-time = "2026-06-05T06:38:45.896Z" }, + { url = "https://files.pythonhosted.org/packages/95/2a/49d8d3e5cef443a11670bae7b66ad1710d1520b1b7378aee7e5d190e819d/qh3-1.9.2-cp313-cp313t-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:f9169c78e3d1924ff376585644938c9679ec28974a95e71e915c8c847a3fb250", size = 2036215, upload-time = "2026-06-05T06:38:47.809Z" }, + { url = "https://files.pythonhosted.org/packages/fd/5e/ed0586992b06055520e4661035d9bd0314804936927e1608fb923862b20d/qh3-1.9.2-cp313-cp313t-manylinux_2_17_ppc64.manylinux2014_ppc64.whl", hash = "sha256:abbc765baa7467f4a67decebe63669675cd1d56f22379ebeb4b0d9e9d8a63169", size = 2027902, upload-time = "2026-06-05T06:38:49.541Z" }, + { url = "https://files.pythonhosted.org/packages/2b/41/3985e1a1d023db3bfcd35920e3f182678388446ef2c382371aa3177a4ae3/qh3-1.9.2-cp313-cp313t-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:7dfc05381d157c82d9b564ce278cf860acc9a0444933b1d0f5000d6497a7dea2", size = 2031368, upload-time = "2026-06-05T06:38:51.243Z" }, + { url = "https://files.pythonhosted.org/packages/4a/35/9deedcb305d12172e95924405d9263ee5fab482b880d569ea6be9ca72517/qh3-1.9.2-cp313-cp313t-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:3405db5eedcae8ffc8e69baf5584e9a2e0cf40a477b171627e9e932fdbf71be5", size = 2085144, upload-time = "2026-06-05T06:38:52.97Z" }, + { url = "https://files.pythonhosted.org/packages/80/e5/3513ae6f452b939c1a00d24726f85fc83d31dec176d046a09c8886b8b72d/qh3-1.9.2-cp313-cp313t-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:585f7a44e69ed108bac674ac95f93c2a8c36358c712c8fd05c4276042f80f9b2", size = 2358308, upload-time = "2026-06-05T06:38:54.819Z" }, + { url = "https://files.pythonhosted.org/packages/c1/78/0a501d5c0992847a70f856ec6ace8cbbcabe690dac9c4f160435cd9be23c/qh3-1.9.2-cp313-cp313t-manylinux_2_39_riscv64.whl", hash = "sha256:5a0ac732582241cdc4063cba4d67cde5b87d5dc183a2a1cabd6e032aa0cf4d6c", size = 2013884, upload-time = "2026-06-05T06:38:56.832Z" }, + { url = "https://files.pythonhosted.org/packages/b9/02/c60019c93051c4564489c70ff13791ca1d573ec834e25d90d5cc5e8dfc26/qh3-1.9.2-cp313-cp313t-musllinux_1_1_aarch64.whl", hash = "sha256:071bb169aa1ce008782c288600bdc566b64c95fa4f0cdab27086493000eaa4c0", size = 2348500, upload-time = "2026-06-05T06:38:58.878Z" }, + { url = "https://files.pythonhosted.org/packages/91/20/e39840f4b514d56494d51f912ef6bac50b62286a0f8f320e87fb7c65ce02/qh3-1.9.2-cp313-cp313t-musllinux_1_1_armv7l.whl", hash = "sha256:896aa258ef5db76861d9a59aef1abf7db9300e33513a116c7894a53dda0487e8", size = 2127127, upload-time = "2026-06-05T06:39:01.012Z" }, + { url = "https://files.pythonhosted.org/packages/c2/4b/fb3f8ae427c1f8731b5feb060f9f509b66297ed658c9fe7de38f95def62a/qh3-1.9.2-cp313-cp313t-musllinux_1_1_i686.whl", hash = "sha256:cc031d94bbcfce9d3c504d60d14e170a51e7115653aabdaf3d6b59e1bacd2c47", size = 2225110, upload-time = "2026-06-05T06:39:03.18Z" }, + { url = "https://files.pythonhosted.org/packages/29/05/468229e513468ff078adfc1f923db8653bea02af3203b5e624f86d1c3376/qh3-1.9.2-cp313-cp313t-musllinux_1_1_riscv64.whl", hash = "sha256:6f454bdf8c613a6e11b70881b830ca3c6d09151e6755e1a88932c4872ad67412", size = 2124317, upload-time = "2026-06-05T06:39:05.296Z" }, + { url = "https://files.pythonhosted.org/packages/7d/ab/88fdc69d53daed0611c9918f186d65869d18088147aed1098fec5e27230e/qh3-1.9.2-cp313-cp313t-musllinux_1_1_x86_64.whl", hash = "sha256:85e48dabe519fc30a58550c2c1e2ec152757eb84d342c76643b1f4034a1d920d", size = 2583504, upload-time = "2026-06-05T06:39:08.341Z" }, + { url = "https://files.pythonhosted.org/packages/78/f1/4709db1a09151e6e0990c8740879c7914f4789082844432428cdaead490e/qh3-1.9.2-cp313-cp313t-win32.whl", hash = "sha256:f54b1abaaed2c36a80f654b3ccde20dd34c369d692a5164cccaf7beed013643e", size = 1862603, upload-time = "2026-06-05T06:39:10.308Z" }, + { url = "https://files.pythonhosted.org/packages/ee/e4/b8d588e42de36f6fdc7c46748175ebce2bfb9810459734ade9e29000296e/qh3-1.9.2-cp313-cp313t-win_amd64.whl", hash = "sha256:6e7491d30f7282440f23be0bffc0f0119a8d7d11f874d3ea03aa7958663cf3c0", size = 2126808, upload-time = "2026-06-05T06:39:12.436Z" }, + { url = "https://files.pythonhosted.org/packages/e3/24/bbb238d90cfe2dc56f51f64c41cb8b17c6751b5978dde93b1e348338a07c/qh3-1.9.2-cp313-cp313t-win_arm64.whl", hash = "sha256:52721659dc4c2fe7e326d72d841807b5569c4eb934e699223970d9c413742d6c", size = 1963385, upload-time = "2026-06-05T06:39:14.274Z" }, + { url = "https://files.pythonhosted.org/packages/e4/93/8cd573bf872dfdba6ca0cace0367f1c6450f1fc399b28e8377f9735e8217/qh3-1.9.2-cp37-abi3-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:aabb7c86f49c668ab5bfb297c5da18755a3ca9cf362f8ce73b86f55f43f662fc", size = 4438775, upload-time = "2026-06-05T06:39:49.072Z" }, + { url = "https://files.pythonhosted.org/packages/70/04/7464f2d5051e08d04fb1288026a556001001e1a9f14f4d7b53e5891b503d/qh3-1.9.2-cp37-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:f7e632d420d88e30e4e3fc9f5457c362e964e93e91f9c90d148b68b0d4b0dc06", size = 2166706, upload-time = "2026-06-05T06:39:50.991Z" }, + { url = "https://files.pythonhosted.org/packages/4f/5e/54a51d257ff9353ea510e8ddd3071c67fa3c3b2b51bd5f549ea9ad034f48/qh3-1.9.2-cp37-abi3-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:b5260ff1c42f7221fb7e92bc9c55a009341c2851f2dbb28475ddcf0a32c2226a", size = 1860460, upload-time = "2026-06-05T06:39:52.784Z" }, + { url = "https://files.pythonhosted.org/packages/cf/bc/753afd386e2ba5c06cd9ff861c674b4991696257d17004325858a9faf0bd/qh3-1.9.2-cp37-abi3-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:304e81b4cf3f97610a1aceaa1aec90a97f0c700071efcc9d6a3b86da0b57539d", size = 2040358, upload-time = "2026-06-05T06:39:54.701Z" }, + { url = "https://files.pythonhosted.org/packages/4a/99/b8a9d7371b4f151a6b168c9b3462ee3c66a8f3db25c78bc26d53fff629c0/qh3-1.9.2-cp37-abi3-manylinux_2_17_ppc64.manylinux2014_ppc64.whl", hash = "sha256:aa20581d3daf8f80b62f42a10c1f69c12f3b6310f141a2e2f4259de588e570fe", size = 2035301, upload-time = "2026-06-05T06:39:56.511Z" }, + { url = "https://files.pythonhosted.org/packages/03/2a/fb12bc4f3c6ff623c76cdd9317d887d694c09433382b7425d441a041f1a5/qh3-1.9.2-cp37-abi3-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:dd3a3d7d737644a4d24117613e89220cd5223cb1e0c8281a5a65a97ebaf1efea", size = 2038507, upload-time = "2026-06-05T06:39:58.313Z" }, + { url = "https://files.pythonhosted.org/packages/7a/f5/202cf706072672d96e5d605862780c259f0e80b95893f8679e1ce466f275/qh3-1.9.2-cp37-abi3-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:82edac5bc2eea0881baa142d93e80469347875909acc19ce5e4236b50c465f8e", size = 2088606, upload-time = "2026-06-05T06:40:00.166Z" }, + { url = "https://files.pythonhosted.org/packages/57/f6/58b48f41024d64afd33162f91f0c757a3d3212b47512a8a53dd3078b604e/qh3-1.9.2-cp37-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:2b18b9d2f892bb95f7f3d3428bb4166f974c9a2e26fcf91b33dda9b4285e7928", size = 2363544, upload-time = "2026-06-05T06:40:01.987Z" }, + { url = "https://files.pythonhosted.org/packages/7a/4d/bf7e8c9763a309a4ab16592229cb4bbd693eefdd4047e7b4c50fd97e38c8/qh3-1.9.2-cp37-abi3-manylinux_2_39_riscv64.whl", hash = "sha256:06f2a10a01b71137fef8bb7b4115f1016d6e4a7f056ce41517ff157e808bf444", size = 2017181, upload-time = "2026-06-05T06:40:03.758Z" }, + { url = "https://files.pythonhosted.org/packages/5c/7a/b3cc3733c52d18d8a38eeaf1e55909b50831b41af7ffe6ee7b71784c9b13/qh3-1.9.2-cp37-abi3-musllinux_1_1_aarch64.whl", hash = "sha256:e15519dc326315c52ab708028bf72aa9162eb3c1bd0ba130dc6d371b000639d2", size = 2355863, upload-time = "2026-06-05T06:40:05.812Z" }, + { url = "https://files.pythonhosted.org/packages/2b/88/d6f46e18d744c48ed302f58586fdb3fd80268537b592421390d362e02962/qh3-1.9.2-cp37-abi3-musllinux_1_1_armv7l.whl", hash = "sha256:7e04bf8bbbfb8a9f363fd38834aa9c52fcc0164182c72ea10cf30a3e2df066c9", size = 2130110, upload-time = "2026-06-05T06:40:07.707Z" }, + { url = "https://files.pythonhosted.org/packages/3a/c2/29096a7633e37e6b414703b37e77179dbfe7dbc732a0c932bf6b786d1d8b/qh3-1.9.2-cp37-abi3-musllinux_1_1_i686.whl", hash = "sha256:e3b0b0d17268d57b808a658db7fd7c5e62561ad6d462949d8466c92b22551067", size = 2229846, upload-time = "2026-06-05T06:40:09.543Z" }, + { url = "https://files.pythonhosted.org/packages/50/1f/7f709f0316d476a732bf17c5489dce33567db058ff3e15478d85da163d9c/qh3-1.9.2-cp37-abi3-musllinux_1_1_riscv64.whl", hash = "sha256:efe5ad0a9df71a9a5881615b0c0888767ad3a7cab05e54ba98db52d89d0b796a", size = 2127267, upload-time = "2026-06-05T06:40:11.636Z" }, + { url = "https://files.pythonhosted.org/packages/d4/03/063ab3975ef6805226de752b89a459d8affb09d2a17760b281db95cc0f07/qh3-1.9.2-cp37-abi3-musllinux_1_1_x86_64.whl", hash = "sha256:742dd7c5e2207db27d24036ca599cad056bd5e58a5ce1b43f0c424df149e17d3", size = 2588750, upload-time = "2026-06-05T06:40:13.667Z" }, + { url = "https://files.pythonhosted.org/packages/79/c7/d007929e0c78e35779fc4da4c6bce56f655e444e268702842a32cc8b2089/qh3-1.9.2-cp37-abi3-win32.whl", hash = "sha256:8d1f185a76578929b0de3dcef135c0cc21d4bff0ef00bd75127d244ba3426125", size = 1872154, upload-time = "2026-06-05T06:40:15.694Z" }, + { url = "https://files.pythonhosted.org/packages/76/56/a96b1c770e22158762823ed399bb2e2041af132c316decd43a5971a63830/qh3-1.9.2-cp37-abi3-win_amd64.whl", hash = "sha256:643e8aa05519dddaf20159d1c13c15473de7649622a1c438968ce0f0da973951", size = 2136158, upload-time = "2026-06-05T06:40:17.502Z" }, + { url = "https://files.pythonhosted.org/packages/40/2e/6c696c9139999f91ffb728eeb2781922d25a9b2e9bd7f370b3f0ac12c31d/qh3-1.9.2-cp37-abi3-win_arm64.whl", hash = "sha256:87bace13780b3c2238db1dd31deb10152203d5b360ca6515c78b853796245606", size = 1973744, upload-time = "2026-06-05T06:40:19.588Z" }, + { url = "https://files.pythonhosted.org/packages/7a/1d/79107d008665b40ac5cb34fdf3d52c7d2c65aa74aba0bd73bb63f593cb84/qh3-1.9.2-pp310-pypy310_pp73-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:9f358490f85f790c06275e967aa0a80f90e064cd494b1e2e7fee190000299be4", size = 4437683, upload-time = "2026-06-05T06:40:22.395Z" }, + { url = "https://files.pythonhosted.org/packages/82/74/065650b4b56c23357cab5f73e51aa8f7dd390e9ad38a525192e4d096c134/qh3-1.9.2-pp310-pypy310_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:0809b09629a5f776112040445634c37ff7ad6d1bb5d6061c0485777c61726ff6", size = 2166050, upload-time = "2026-06-05T06:40:25.059Z" }, + { url = "https://files.pythonhosted.org/packages/6c/11/2ae0837e58d831a11c21a15041f3ed67f63607f043dab987b8e09845ddf0/qh3-1.9.2-pp310-pypy310_pp73-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:64f676dd7292c77ec122a2b9dd491fad5a0adb7e56b7db4f7e0306ada3664e88", size = 1860395, upload-time = "2026-06-05T06:40:26.844Z" }, + { url = "https://files.pythonhosted.org/packages/1e/f6/d3f927755ae1165c6a945f88ea9da85cdc5eb5feb5eb19ade572cb6228c5/qh3-1.9.2-pp310-pypy310_pp73-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:72d38eeaf6d6ae6219b5947f7273c02d681271b6549f581ff7da33d5b8e05f57", size = 2040044, upload-time = "2026-06-05T06:40:28.927Z" }, + { url = "https://files.pythonhosted.org/packages/14/4c/afc4959b3fb233d5d72ce0063c4c26c66489c4ce63bd4fa431ef9626ffc8/qh3-1.9.2-pp310-pypy310_pp73-manylinux_2_17_ppc64.manylinux2014_ppc64.whl", hash = "sha256:54749bc37813cdf85e02235feced91f4d421e5be118524b225087eb04482dbe5", size = 2034044, upload-time = "2026-06-05T06:40:30.982Z" }, + { url = "https://files.pythonhosted.org/packages/a8/74/0b10bffeb460b69f6c6629941c559038163bee391f422480cdde4f08c507/qh3-1.9.2-pp310-pypy310_pp73-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:286df0555c823f88c6899f8d18d21f85b2d666f22e35950bcb4c7f25df1c0209", size = 2037363, upload-time = "2026-06-05T06:40:32.906Z" }, + { url = "https://files.pythonhosted.org/packages/5f/3d/a0377632cb518801cdda7c2e76505e06a2801d330e1356bcfc4f23713194/qh3-1.9.2-pp310-pypy310_pp73-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:f67694450684bdc3751f16134ae8a6ffc113e023c8c159d537d15ef47ab752fe", size = 2087820, upload-time = "2026-06-05T06:40:34.97Z" }, + { url = "https://files.pythonhosted.org/packages/ed/69/39e96d47030475a708abccb046eb1e9185e5cfc4e1148d7e36a79eb8fd46/qh3-1.9.2-pp310-pypy310_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:3da0ce784c43dba22fb24221559a10c4652f3d75e8bd0990d8e9fb27b3815b7a", size = 2362658, upload-time = "2026-06-05T06:40:38.066Z" }, + { url = "https://files.pythonhosted.org/packages/25/ef/a8ca5e3850ca045cb8e8e72d48d85c3f6c6c1c5a3b42a83347e57ccca4db/qh3-1.9.2-pp310-pypy310_pp73-musllinux_1_1_aarch64.whl", hash = "sha256:d4c66ae0fdc8d1b9e391d362d08d16553e6997556724e8106571f943597b93aa", size = 2355136, upload-time = "2026-06-05T06:40:40.288Z" }, + { url = "https://files.pythonhosted.org/packages/51/75/2d97d18dec97fff7d2adbc3b06fc4456c24cd2cc3d3253d2076679b3745b/qh3-1.9.2-pp310-pypy310_pp73-musllinux_1_1_armv7l.whl", hash = "sha256:4577bb73f6cd906778a5e7d0146a31f67c49aa08d3b33452f7081438ee739446", size = 2130356, upload-time = "2026-06-05T06:40:42.27Z" }, + { url = "https://files.pythonhosted.org/packages/2a/9e/276c72c9e97d226211761929dffdc0a72a60ee6313ae0573603541e47e02/qh3-1.9.2-pp310-pypy310_pp73-musllinux_1_1_i686.whl", hash = "sha256:77b46285a74fa49c34a51df690c8a31f3df290950b552ed382811808c1179c4e", size = 2229395, upload-time = "2026-06-05T06:40:44.349Z" }, + { url = "https://files.pythonhosted.org/packages/c2/0c/15c12233b0391ad65aeb4fe835cceed60429b9de2ca9e4f15dbb0fd4fe7f/qh3-1.9.2-pp310-pypy310_pp73-musllinux_1_1_x86_64.whl", hash = "sha256:da0a41fedc281a226e234d4cafd82dd7b6ec5b9546fa25d3fa3e3c9ba6bbacb9", size = 2588194, upload-time = "2026-06-05T06:40:46.75Z" }, + { url = "https://files.pythonhosted.org/packages/1b/53/f84d9b3507b9b609f707ab48a37b665cb84638f84254548b1fd5c6948123/qh3-1.9.2-pp310-pypy310_pp73-win_amd64.whl", hash = "sha256:8aae8ed6bff1256700abe4037b8098645587cbdd17b33d8a674d2fb615f5aef2", size = 2133546, upload-time = "2026-06-05T06:40:48.731Z" }, + { url = "https://files.pythonhosted.org/packages/ec/96/52a06a8211a23742997dc3c81473e1f0fa6045b0605dce35cda51c352758/qh3-1.9.2-pp311-pypy311_pp73-macosx_10_12_x86_64.macosx_11_0_arm64.macosx_10_12_universal2.whl", hash = "sha256:b193dfae9c24b2329ad53871a8b973239010293f60bdee76d43ce2f38d54a9bf", size = 4433993, upload-time = "2026-06-05T06:40:50.908Z" }, + { url = "https://files.pythonhosted.org/packages/b4/d4/c1c7c48a5c3e3b2e3502e0b678d3b9d26a01e14d985b87c506057c282b32/qh3-1.9.2-pp311-pypy311_pp73-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:ed3b9086441bec662c507d960fe5aaeed346eaef3a57555b95ed8f3e1e44f8f2", size = 2162613, upload-time = "2026-06-05T06:40:53.121Z" }, + { url = "https://files.pythonhosted.org/packages/78/c7/f91abcab1f325f4b3265e414e9e19138a975f817bfecafe4a46e9c7de3ba/qh3-1.9.2-pp311-pypy311_pp73-manylinux_2_17_armv7l.manylinux2014_armv7l.whl", hash = "sha256:a69d017995e64df1679c25e7bf30fd134550872ba10fb4dfb078a6e514117977", size = 1858138, upload-time = "2026-06-05T06:40:55.174Z" }, + { url = "https://files.pythonhosted.org/packages/df/38/ab704e632451f2d9251ed665e206bfbae2a474eb9ed6f81d71f217c4e96a/qh3-1.9.2-pp311-pypy311_pp73-manylinux_2_17_i686.manylinux2014_i686.whl", hash = "sha256:e030701334043a9a5b1cb54ccd9392cbcb1de5f11e34ac4e90b2f940fe66b3e0", size = 2038541, upload-time = "2026-06-05T06:40:57.189Z" }, + { url = "https://files.pythonhosted.org/packages/8a/bf/1be17524bec6ca8ecec0b0822aa561890f9d8f862a3e034fcc0dc9797dfd/qh3-1.9.2-pp311-pypy311_pp73-manylinux_2_17_ppc64.manylinux2014_ppc64.whl", hash = "sha256:6f6e41bc5639fa009760ce2bcabe9aacf89dff6532dffa02fc1f94de090f1968", size = 2031619, upload-time = "2026-06-05T06:40:59.093Z" }, + { url = "https://files.pythonhosted.org/packages/2b/eb/828ba994c38eab2e91e9d8fc6b0741ad7c70d3b9d229b069495e02e42639/qh3-1.9.2-pp311-pypy311_pp73-manylinux_2_17_ppc64le.manylinux2014_ppc64le.whl", hash = "sha256:564ab86fca8b2cf9e109189ffb4c527e1fc4466f675d5709409ed0e419dda42f", size = 2034383, upload-time = "2026-06-05T06:41:00.954Z" }, + { url = "https://files.pythonhosted.org/packages/cb/9d/50888da0880749419d9d6edefa0654dec167aed87baa10f2107c0be8a14b/qh3-1.9.2-pp311-pypy311_pp73-manylinux_2_17_s390x.manylinux2014_s390x.whl", hash = "sha256:a29247ac022ac2d647bf069ffeaf879abdf0e8d6f24094296027e823473dfb34", size = 2083750, upload-time = "2026-06-05T06:41:03.325Z" }, + { url = "https://files.pythonhosted.org/packages/51/99/1d5d4c3c657763266de93296ac50ac021e6b967d4cc751d144ce3f104c68/qh3-1.9.2-pp311-pypy311_pp73-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:140282ec907be1a16db9a7d59d60e94c35571463d9ba6a92401073e457a03bfa", size = 2359579, upload-time = "2026-06-05T06:41:05.859Z" }, + { url = "https://files.pythonhosted.org/packages/f3/ba/f3a34435a5d3d7bee8babecbf8487696f5ca119ee173941e70c2c99e7b4d/qh3-1.9.2-pp311-pypy311_pp73-musllinux_1_1_aarch64.whl", hash = "sha256:21c4130aa2fc31cabf6b92e3ec3d2ed641d9689ebe73f23d6a947bdd4c9cc2f1", size = 2350846, upload-time = "2026-06-05T06:41:08.198Z" }, + { url = "https://files.pythonhosted.org/packages/6d/10/dd76cbe362b6f7d3b97399eebfd6903640cb67c51a39ae5c9df1b45c3dfa/qh3-1.9.2-pp311-pypy311_pp73-musllinux_1_1_armv7l.whl", hash = "sha256:dd574406736ab11d33a2a972c9e6d5277ccd72b5adcebd3052daf2d5304d4693", size = 2127918, upload-time = "2026-06-05T06:41:10.275Z" }, + { url = "https://files.pythonhosted.org/packages/c2/00/368356c118bb15f3e7fc76de0fb87ae5d41eeb2ad8b9dc1a3528c5169dc1/qh3-1.9.2-pp311-pypy311_pp73-musllinux_1_1_i686.whl", hash = "sha256:521ee6f58d3b802a1e8259d8a9c4cc0027101f2ca6031ba0630a93cad6f16e0d", size = 2227195, upload-time = "2026-06-05T06:41:12.21Z" }, + { url = "https://files.pythonhosted.org/packages/d5/f8/5b97a46ac01653bf136c61927e7617927c4cc2530ada7eef652a3fd3b0b1/qh3-1.9.2-pp311-pypy311_pp73-musllinux_1_1_x86_64.whl", hash = "sha256:41c86993834451502a248bef2362199323a8b2ae943e1cccb415940ca514e601", size = 2585100, upload-time = "2026-06-05T06:41:14.118Z" }, + { url = "https://files.pythonhosted.org/packages/70/9c/4a2c3dd8bd16a7b9df2033be7266d550829ba74cd3172e164ae05dc1806f/qh3-1.9.2-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:6169dfc8322ff1a27179281aad7b36072dee8c8b24843766186a932f6f04d5b0", size = 2131071, upload-time = "2026-06-05T06:41:16.171Z" }, +] + [[package]] name = "referencing" version = "0.37.0" @@ -2434,6 +2627,23 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/a0/f4/c67b0b3f1b9245e8d266f0f112c500d50e5b4e83cb6f3b71b6528104182a/requests-2.34.2-py3-none-any.whl", hash = "sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0", size = 73075, upload-time = "2026-05-14T19:25:26.443Z" }, ] +[[package]] +name = "requests-cache" +version = "1.3.2" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "attrs" }, + { name = "cattrs" }, + { name = "platformdirs" }, + { name = "requests" }, + { name = "url-normalize" }, + { name = "urllib3" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/c3/ae/90a0f931c7f6b5a674b98c25ecb2593a173bcee14f0d8c148471df3d7b26/requests_cache-1.3.2.tar.gz", hash = "sha256:bdc3680931f98a1dea509d339ea6b45cea526945b47b250ce63ffd2744ee0b14", size = 100167, upload-time = "2026-05-11T04:09:53.233Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b0/ff/d87d1a7700463afc5440bec80cfbcb56ef929f05fbfdc946ce031b13d040/requests_cache-1.3.2-py3-none-any.whl", hash = "sha256:c52666c76b08daa94d05a99327dd24afc46f405abc044e8c2267b540f90673d0", size = 70633, upload-time = "2026-05-11T04:09:51.554Z" }, +] + [[package]] name = "res-wind-up" version = "0.5.0" @@ -2458,66 +2668,111 @@ dependencies = [ ] [package.optional-dependencies] +era5 = [ + { name = "openmeteo-requests" }, + { name = "requests-cache" }, + { name = "retry-requests" }, +] +examples = [ + { name = "ephem" }, + { name = "flaml", extra = ["automl"] }, + { name = "ipywidgets" }, + { name = "jupyterlab" }, + { name = "notebook" }, + { name = "requests" }, +] +ml = [ + { name = "lightgbm" }, + { name = "scikit-learn", version = "1.7.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, + { name = "scikit-learn", version = "1.9.0", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.11'" }, +] + +[package.dev-dependencies] dev = [ { name = "coverage" }, + { name = "lightgbm" }, { name = "mypy" }, + { name = "openmeteo-requests" }, { name = "poethepoet" }, { name = "pytest" }, { name = "pytest-env" }, { name = "requests" }, + { name = "requests-cache" }, + { name = "retry-requests" }, { name = "ruff" }, + { name = "scikit-learn", version = "1.7.2", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.11'" }, + { name = "scikit-learn", version = "1.9.0", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version >= '3.11'" }, { name = "types-pyyaml" }, { name = "types-requests" }, { name = "types-tabulate" }, { name = "types-toml" }, { name = "types-tqdm" }, ] -examples = [ - { name = "ephem" }, - { name = "flaml", extra = ["automl"] }, - { name = "ipywidgets" }, - { name = "jupyterlab" }, - { name = "notebook" }, - { name = "requests" }, -] [package.metadata] requires-dist = [ - { name = "coverage", marker = "extra == 'dev'" }, { name = "ephem", marker = "extra == 'examples'" }, { name = "eval-type-backport" }, { name = "flaml", extras = ["automl"], marker = "extra == 'examples'" }, { name = "geographiclib" }, { name = "ipywidgets", marker = "extra == 'examples'" }, { name = "jupyterlab", marker = "extra == 'examples'" }, + { name = "lightgbm", marker = "extra == 'ml'" }, { name = "matplotlib" }, - { name = "mypy", marker = "extra == 'dev'", specifier = "<1.19" }, { name = "notebook", marker = "extra == 'examples'" }, + { name = "openmeteo-requests", marker = "extra == 'era5'" }, { name = "pandas", specifier = ">=2.0.0,<3.0.0" }, - { name = "poethepoet", marker = "extra == 'dev'" }, { name = "pyarrow" }, { name = "pydantic", specifier = ">=2.0.0" }, - { name = "pytest", marker = "extra == 'dev'" }, - { name = "pytest-env", marker = "extra == 'dev'" }, { name = "python-dotenv" }, { name = "pyyaml" }, - { name = "requests", marker = "extra == 'dev'" }, { name = "requests", marker = "extra == 'examples'" }, - { name = "ruff", marker = "extra == 'dev'" }, + { name = "requests-cache", marker = "extra == 'era5'" }, + { name = "retry-requests", marker = "extra == 'era5'" }, { name = "ruptures" }, + { name = "scikit-learn", marker = "extra == 'ml'" }, { name = "scipy" }, { name = "seaborn" }, { name = "tabulate" }, { name = "toml" }, { name = "tqdm" }, - { name = "types-pyyaml", marker = "extra == 'dev'" }, - { name = "types-requests", marker = "extra == 'dev'" }, - { name = "types-tabulate", marker = "extra == 'dev'" }, - { name = "types-toml", marker = "extra == 'dev'" }, - { name = "types-tqdm", marker = "extra == 'dev'" }, { name = "utm" }, ] -provides-extras = ["dev", "examples"] +provides-extras = ["era5", "ml", "examples"] + +[package.metadata.requires-dev] +dev = [ + { name = "coverage" }, + { name = "lightgbm" }, + { name = "mypy", specifier = "<1.19" }, + { name = "openmeteo-requests" }, + { name = "poethepoet" }, + { name = "pytest" }, + { name = "pytest-env" }, + { name = "requests" }, + { name = "requests-cache" }, + { name = "retry-requests" }, + { name = "ruff" }, + { name = "scikit-learn" }, + { name = "types-pyyaml" }, + { name = "types-requests" }, + { name = "types-tabulate" }, + { name = "types-toml" }, + { name = "types-tqdm" }, +] + +[[package]] +name = "retry-requests" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "requests" }, + { name = "urllib3" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/1e/da/6e961557733660bef8d095a1d81423a3707486e2b2ecd2c5ad5ad8d2f59d/retry-requests-2.0.0.tar.gz", hash = "sha256:3d02135e5aafedf09240414182fc7389c5d2b4de0252daba0054c9d6a27e7639", size = 16084, upload-time = "2023-05-28T18:33:08.196Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/b1/f3/8ce908497bebbc2790ef06240a2c0fb28c096abb59062d88f85090464a5f/retry_requests-2.0.0-py3-none-any.whl", hash = "sha256:38e8e3f55051e7b7915c1768884269097865a5da2ea87d5dcafd6ba9498c363f", size = 15772, upload-time = "2023-05-28T18:33:05.868Z" }, +] [[package]] name = "rfc3339-validator" @@ -3265,6 +3520,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/e7/00/3fca040d7cf8a32776d3d81a00c8ee7457e00f80c649f1e4a863c8321ae9/uri_template-1.3.0-py3-none-any.whl", hash = "sha256:a44a133ea12d44a0c0f06d7d42a52d71282e77e2f937d8abd5655b8d56fc1363", size = 11140, upload-time = "2023-06-21T01:49:03.467Z" }, ] +[[package]] +name = "url-normalize" +version = "3.0.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "idna" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/8b/cd/846d87d6d49d963b04ef4429b73d71d3c17468059956bab360866a9b0aec/url_normalize-3.0.0.tar.gz", hash = "sha256:0552cbf2831a32a28994a13d29bca58a60e10ff6c0380e343ec6d1c2a0d232d8", size = 21777, upload-time = "2026-04-25T00:31:59.514Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/13/8a/f72344eab18674fd7b174f35abbce41ed88fea72927f111726732d0ca779/url_normalize-3.0.0-py3-none-any.whl", hash = "sha256:95234bd359f86831c1fd87c248877f2a6887db2f3b5087120083f2fffcba4889", size = 16854, upload-time = "2026-04-25T00:31:58.271Z" }, +] + [[package]] name = "urllib3" version = "2.7.0" @@ -3274,6 +3541,20 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/7f/3e/5db95bcf282c52709639744ca2a8b149baccf648e39c8cc87553df9eae0c/urllib3-2.7.0-py3-none-any.whl", hash = "sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897", size = 131087, upload-time = "2026-05-07T16:13:17.151Z" }, ] +[[package]] +name = "urllib3-future" +version = "2.21.902" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "h11" }, + { name = "jh2" }, + { name = "qh3", marker = "(python_full_version < '3.12' and platform_machine == 'AMD64' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'ARM64' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'aarch64' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'arm64' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'armv7l' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'i686' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'ppc64' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'ppc64le' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'riscv64' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'riscv64gc' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 's390x' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'x86' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'x86_64' and platform_python_implementation == 'PyPy' and sys_platform == 'darwin') or (python_full_version < '3.12' and platform_machine == 'AMD64' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'ARM64' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'aarch64' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'arm64' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'armv7l' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'i686' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'ppc64' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'ppc64le' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'riscv64' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'riscv64gc' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 's390x' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'x86' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'x86_64' and platform_python_implementation == 'PyPy' and sys_platform == 'linux') or (python_full_version < '3.12' and platform_machine == 'AMD64' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'ARM64' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'aarch64' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'arm64' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'armv7l' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'i686' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'ppc64' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'ppc64le' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'riscv64' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'riscv64gc' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 's390x' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'x86' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (python_full_version < '3.12' and platform_machine == 'x86_64' and platform_python_implementation == 'PyPy' and sys_platform == 'win32') or (platform_machine == 'AMD64' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'ARM64' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'aarch64' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'arm64' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'armv7l' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'i686' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'ppc64' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'ppc64le' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'riscv64' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'riscv64gc' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 's390x' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'x86' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'x86_64' and platform_python_implementation == 'CPython' and sys_platform == 'darwin') or (platform_machine == 'AMD64' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'ARM64' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'aarch64' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'arm64' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'armv7l' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'i686' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'ppc64' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'ppc64le' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'riscv64' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'riscv64gc' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 's390x' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'x86' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'x86_64' and platform_python_implementation == 'CPython' and sys_platform == 'linux') or (platform_machine == 'AMD64' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'ARM64' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'aarch64' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'arm64' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'armv7l' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'i686' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'ppc64' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'ppc64le' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'riscv64' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'riscv64gc' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 's390x' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'x86' and platform_python_implementation == 'CPython' and sys_platform == 'win32') or (platform_machine == 'x86_64' and platform_python_implementation == 'CPython' and sys_platform == 'win32')" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/8c/9e/43a5fbd5af6dd61fc6a64f8a0d1e190c5b335fd6b2442aa30bc37306d6b7/urllib3_future-2.21.902.tar.gz", hash = "sha256:9a1a9d600394e73c65057dfa26e30de93beea879ea8d17e8003e130bf78368f6", size = 1299740, upload-time = "2026-06-01T12:03:33.43Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/5e/0e/e50386b53f5ae135147e5762a16490e521a85814dc9db9a4d091ff22821d/urllib3_future-2.21.902-py3-none-any.whl", hash = "sha256:0e7f57858b9faf12bf84f30ff86ca9fffeb271f8bd92fd519b765a89c46f4962", size = 772463, upload-time = "2026-06-01T12:03:31.402Z" }, +] + [[package]] name = "utm" version = "0.8.1" @@ -3283,6 +3564,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/3b/a4/0698f3e5c397442ec9323a537e48cc63b846288b6878d38efd04e91005e3/utm-0.8.1-py3-none-any.whl", hash = "sha256:e3d5e224082af138e40851dcaad08d7f99da1cc4b5c413a7de34eabee35f434a", size = 8613, upload-time = "2025-03-06T11:40:54.273Z" }, ] +[[package]] +name = "wassima" +version = "2.1.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b8/34/68ab01470c1cef170e8370a8a05e598d621d3657bf925b62bc9a18b4509a/wassima-2.1.1.tar.gz", hash = "sha256:9c6ad4aa3cfbe91fd75f9eae315ba563bbc7d9d2479aef0c288fa7f1ca3b0c53", size = 140395, upload-time = "2026-06-08T02:44:41.424Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/54/9e/472991fc66d940d3ba3268fc8c2866a44b6870a9711ab74fcf1f29e817e1/wassima-2.1.1-py3-none-any.whl", hash = "sha256:ab0f12c091ff697f111eea5f925aafd83736909a1c9c764a8bbf874b2e4d4a42", size = 131435, upload-time = "2026-06-08T02:44:39.92Z" }, +] + [[package]] name = "wcwidth" version = "0.8.1" diff --git a/wind_up/era5.py b/wind_up/era5.py new file mode 100644 index 00000000..9f7b71f6 --- /dev/null +++ b/wind_up/era5.py @@ -0,0 +1,172 @@ +"""ERA5 reanalysis data fetching via the Open-Meteo archive API. + +The Open-Meteo client libraries (``openmeteo_requests``, ``requests_cache``, ``retry_requests``) +are an optional dependency group (``era5``) and are imported lazily, so this module imports +without them; only the live network fetch needs them installed. + +The returned DataFrame uses the Open-Meteo column names (``wind_speed_100m`` / +``wind_direction_100m``); :func:`wind_up.reanalysis_data._reanalysis_upsample` already renames +those to the wind-up reanalysis columns, so a :class:`~wind_up.reanalysis_data.ReanalysisDataset` +built from it drops straight into an analysis. +""" + +from __future__ import annotations + +import hashlib +import json +import logging +import os +from pathlib import Path + +import pandas as pd + +logger = logging.getLogger(__name__) + +CACHE_DIR_ENV = "WIND_UP_CACHE_DIR" + +ERA5_DEFAULT_FIELDS: list[str] = [ + "temperature_2m", + "relative_humidity_2m", + "dew_point_2m", + "apparent_temperature", + "pressure_msl", + "surface_pressure", + "precipitation", + "rain", + "snowfall", + "cloud_cover", + "shortwave_radiation", + "direct_radiation", + "diffuse_radiation", + "wind_speed_10m", + "wind_speed_100m", + "wind_direction_10m", + "wind_direction_100m", + "wind_gusts_10m", + "weather_code", +] + + +def _resolve_cache_dir(cache_dir: str | Path | None) -> Path: + """Resolve the wind-up cache directory. + + Precedence: explicit ``cache_dir`` arg, then the ``WIND_UP_CACHE_DIR`` env var, then + ``~/.cache/wind_up``. The directory is not created here (callers create it before writing). + """ + if cache_dir is not None: + return Path(cache_dir) + env = os.getenv(CACHE_DIR_ENV) + if env: + return Path(env) + return Path.home() / ".cache" / "wind_up" + + +def _build_era5_df(response: object, fields: list[str]) -> pd.DataFrame: + """Build a tidy hourly DataFrame from a single Open-Meteo response object.""" + hourly_data = response.Hourly() # type: ignore[attr-defined] # openmeteo_requests has no type stubs + return pd.DataFrame( + { + "timestamp": pd.date_range( + start=pd.to_datetime(hourly_data.Time(), unit="s", utc=True), + end=pd.to_datetime(hourly_data.TimeEnd(), unit="s", utc=True), + freq=pd.Timedelta(seconds=hourly_data.Interval()), + inclusive="left", + ) + } + | {field: hourly_data.Variables(i).ValuesAsNumpy() for i, field in enumerate(fields)} + ).set_index("timestamp") + + +def _era5_cache_path( + lat: float, + lon: float, + start_date: str, + end_date: str, + fields: list[str], + *, + cache_dir: str | Path | None = None, +) -> Path: + """Build a deterministic parquet cache path from the request args.""" + args_blob = json.dumps( + {"lat": lat, "lon": lon, "start_date": start_date, "end_date": end_date, "fields": list(fields)}, + sort_keys=True, + ) + args_hash = hashlib.sha256(args_blob.encode("utf-8")).hexdigest()[:16] + base = _resolve_cache_dir(cache_dir) / "era5_data" + return base / f"ERA5_{lat:.2f}_{lon:.2f}_{start_date}_{end_date}_{args_hash}.parquet" + + +def get_era5_hourly_df( + *, + lat: float, + lon: float, + start_date: str = "2000-01-01", + end_date: str | None = None, + fields: list[str] | None = None, + cache_dir: str | Path | None = None, +) -> pd.DataFrame: + """Fetch hourly ERA5 data from Open-Meteo for any location and return as a DataFrame. + + ``fields`` defaults to a copy of :data:`ERA5_DEFAULT_FIELDS` when ``None``. ``end_date`` + defaults to today (UTC) when ``None``. Each unique combination of arguments is cached to + its own parquet file keyed by a hash of the arguments. Delete the cache file to force a + refetch. + """ + if fields is None: + fields = list(ERA5_DEFAULT_FIELDS) + if end_date is None: + end_date = pd.Timestamp.now(tz="UTC").normalize().strftime("%Y-%m-%d") + cache_path = _era5_cache_path(lat, lon, start_date, end_date, fields, cache_dir=cache_dir) + if cache_path.exists(): + logger.info("Reading: %s", cache_path) + return pd.read_parquet(cache_path) + + df = _fetch_era5_from_open_meteo( # pragma: no cover - live network fetch + lat=lat, lon=lon, start_date=start_date, end_date=end_date, fields=fields, cache_dir=cache_dir + ) + cache_path.parent.mkdir(parents=True, exist_ok=True) # pragma: no cover - live network fetch + logger.info("Writing: %s", cache_path) # pragma: no cover - live network fetch + df.to_parquet(cache_path) # pragma: no cover - live network fetch + return df # pragma: no cover - live network fetch + + +def _fetch_era5_from_open_meteo( # pragma: no cover - live network fetch + *, + lat: float, + lon: float, + start_date: str, + end_date: str, + fields: list[str], + cache_dir: str | Path | None, +) -> pd.DataFrame: + """Fetch a single Open-Meteo archive response and build the hourly DataFrame. + + The Open-Meteo client libraries are imported here so the module imports without the + optional ``era5`` dependency group installed. + """ + import openmeteo_requests # noqa: PLC0415 + import requests_cache # noqa: PLC0415 + from retry_requests import retry # noqa: PLC0415 + + requests_cache_path = _resolve_cache_dir(cache_dir) / "openmeteo_requests_cache" + requests_cache_path.parent.mkdir(parents=True, exist_ok=True) + openmeteo = openmeteo_requests.Client( + session=retry( + requests_cache.CachedSession(str(requests_cache_path), expire_after=3600), + retries=5, + backoff_factor=0.2, + ) + ) + responses = openmeteo.weather_api( + url="https://archive-api.open-meteo.com/v1/archive", + params={ + "latitude": lat, + "longitude": lon, + "start_date": start_date, + "end_date": end_date, + "hourly": fields, + "models": "era5", + "wind_speed_unit": "ms", + }, + ) + return _build_era5_df(responses[0], fields) diff --git a/wind_up/main_analysis.py b/wind_up/main_analysis.py index 91ee8aff..c27dc5ad 100644 --- a/wind_up/main_analysis.py +++ b/wind_up/main_analysis.py @@ -296,8 +296,8 @@ def _toggle_pairing_filter( else: msg = f"pairing_filter_method {pairing_filter_method} not recognised" raise ValueError(msg) - len_pre_after = len(filt_pre_df) - len_post_after = len(filt_post_df) + len_pre_after = len(filt_pre_df.dropna(subset=required_cols)) + len_post_after = len(filt_post_df.dropna(subset=required_cols)) logger.info( f"removed {len_pre_before - len_pre_after} [{100 * (len_pre_before - len_pre_after) / len_pre_before:.1f}%] " f"rows from pre_df using {pairing_filter_method} pairing filter", diff --git a/wind_up/plots/northing_plots.py b/wind_up/plots/northing_plots.py index a313e67b..7bc8bcdc 100644 --- a/wind_up/plots/northing_plots.py +++ b/wind_up/plots/northing_plots.py @@ -127,6 +127,9 @@ def plot_northing_changepoint( plt.legend(fontsize="small", ncol=2) plt.grid() plt.tight_layout() - (plot_cfg.plots_dir / northing_turbine).mkdir(exist_ok=True) - plt.savefig(plot_cfg.plots_dir / northing_turbine / f"{title}.png") + if plot_cfg.show_plots: + plt.show() + if plot_cfg.save_plots: + (plot_cfg.plots_dir / northing_turbine).mkdir(parents=True, exist_ok=True) + plt.savefig(plot_cfg.plots_dir / northing_turbine / f"{title}.png") plt.close() diff --git a/wind_up/plots/optimize_northing_plots.py b/wind_up/plots/optimize_northing_plots.py index 8004053a..705f8be6 100644 --- a/wind_up/plots/optimize_northing_plots.py +++ b/wind_up/plots/optimize_northing_plots.py @@ -33,7 +33,11 @@ def plot_diff_to_north_ref_wd( plt.xlabel("datetime") plt.ylabel(f"yaw angle diff to {north_ref_wd_col} [deg]") plt.tight_layout() - plt.savefig(plot_cfg.plots_dir / wtg_name / f"{title}.png") + if plot_cfg.show_plots: + plt.show() + if plot_cfg.save_plots: + (plot_cfg.plots_dir / wtg_name).mkdir(exist_ok=True, parents=True) + plt.savefig(plot_cfg.plots_dir / wtg_name / f"{title}.png") plt.close() @@ -45,8 +49,11 @@ def plot_yaw_diff_vs_power(wtg_df: pd.DataFrame, *, wtg_name: str, north_ref_wd_ plt.ylabel(f"yaw_diff_to_{north_ref_wd_col}") title = f"{wtg_name} yaw_diff_to_{north_ref_wd_col} vs {RAW_POWER_COL}" plt.tight_layout() - (plot_cfg.plots_dir / wtg_name).mkdir(exist_ok=True, parents=True) - plt.savefig(plot_cfg.plots_dir / wtg_name / f"{title}.png") + if plot_cfg.show_plots: + plt.show() + if plot_cfg.save_plots: + (plot_cfg.plots_dir / wtg_name).mkdir(exist_ok=True, parents=True) + plt.savefig(plot_cfg.plots_dir / wtg_name / f"{title}.png") plt.close() @@ -67,5 +74,9 @@ def plot_wf_yawdir_and_reanalysis_timeseries(wf_df: pd.DataFrame, *, cfg: WindUp plt.legend() plt.grid() plt.tight_layout() - plt.savefig(plot_cfg.plots_dir / f"{title}.png") + if plot_cfg.show_plots: + plt.show() + if plot_cfg.save_plots: + plot_cfg.plots_dir.mkdir(exist_ok=True, parents=True) + plt.savefig(plot_cfg.plots_dir / f"{title}.png") plt.close() diff --git a/wind_up/plots/scada_funcs_plots.py b/wind_up/plots/scada_funcs_plots.py index 1bf56d4f..77a231aa 100644 --- a/wind_up/plots/scada_funcs_plots.py +++ b/wind_up/plots/scada_funcs_plots.py @@ -57,13 +57,16 @@ def calc_cf_by_turbine(scada_df: pd.DataFrame, cfg: WindUpConfig) -> pd.DataFram def print_and_plot_capacity_factor(scada_df: pd.DataFrame, cfg: WindUpConfig, plots_cfg: PlotConfig) -> None: cf_df = calc_cf_by_turbine(scada_df=scada_df, cfg=cfg) title = f"{cfg.asset.name} capacity factor" - plots_cfg.plots_dir.mkdir(parents=True, exist_ok=True) + save_path = None + if plots_cfg.save_plots: + plots_cfg.plots_dir.mkdir(parents=True, exist_ok=True) + save_path = plots_cfg.plots_dir / f"{title}.png" bubble_plot( cfg=cfg, series=cf_df["CF"] * 100, title=f"{cfg.asset.name} capacity factor", cbarunits="%", - save_path=plots_cfg.plots_dir / f"{title}.png", + save_path=save_path, show_plot=plots_cfg.show_plots, )