From 88a7bac077cd6609e943f7ef8a50aed29c4841ac Mon Sep 17 00:00:00 2001 From: DONNOT Benjamin Date: Fri, 11 Sep 2026 15:35:02 +0200 Subject: [PATCH 1/7] improve the ml powerflow operator Assisted-by: Claude Code (Sonnet 5) Signed-off-by: DONNOT Benjamin --- CLAUDE.md | 10 +- README.md | 22 +- docs/api.rst | 151 ++++++ docs/conf.py | 2 +- docs/examples.rst | 13 + examples/README.md | 1 + pyproject.toml | 5 + src/_cpp/acpf_nr_kernels.cuh | 101 ++++ src/_cpp/contingency/batch_pf_driver.cu | 322 +++++++++++ src/_cpp/contingency/batch_pf_driver.cuh | 181 +++++++ .../batch_sources/scenario_sweep_batch.cu | 78 ++- .../batch_sources/scenario_sweep_batch.cuh | 188 ++++--- src/_cpp/contingency/driver.cuh | 23 +- src/_cpp/dlpack_export.cu | 212 ++++++++ src/_cpp/dlpack_export.cuh | 9 +- src/_cpp/dlpack_export.hpp | 46 ++ src/_cpp/nr_iter_step.cuh | 40 +- src/_cpp/python_bindings.cpp | 137 ++++- src/_cpp/scenario_sweep_session.cu | 462 +++++++++++++--- src/_cpp/scenario_sweep_session.hpp | 166 +++++- src/_cpp/timing_utils.hpp | 13 + src/gpusim2grid/__init__.py | 2 +- src/gpusim2grid/_ls2g_utils.py | 16 +- src/gpusim2grid/differentiable/__init__.py | 3 +- src/gpusim2grid/differentiable/_batch_pf.py | 472 ++++++++++++++++ src/gpusim2grid/differentiable/_flows.py | 16 +- src/gpusim2grid/scenario_sweep/__init__.py | 132 ++++- src/gpusim2grid/scenario_sweep/gpu_facade.py | 54 +- tests/python/test_batch_power_flow.py | 507 ++++++++++++++++++ .../python/test_scenario_sweep_persistence.py | 351 ++++++++++++ 30 files changed, 3508 insertions(+), 227 deletions(-) create mode 100644 src/gpusim2grid/differentiable/_batch_pf.py create mode 100644 tests/python/test_batch_power_flow.py create mode 100644 tests/python/test_scenario_sweep_persistence.py diff --git a/CLAUDE.md b/CLAUDE.md index 664a34a..2be5b18 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -148,12 +148,20 @@ Three construction-time-only knobs select cuDSS's analysis-phase behavior, expos ### Zero-copy interop (DLPack) -`dlpack_export.{cu,cuh,hpp}` exports device voltage buffers as DLPack capsules (`v_base_dlpack()`, `v_results_dlpack()`) for zero-copy `torch.from_dlpack()` / `jax.dlpack.from_dlpack()`. **These capsules alias live GPU memory** — a subsequent `run()` overwrites them in place; clone the tensor if you need a snapshot, and keep the session object alive while the tensor is in use. +`dlpack_export.{cu,cuh,hpp}` exports device voltage buffers as DLPack capsules (`v_base_dlpack()`, `v_results_dlpack()`) for zero-copy `torch.from_dlpack()` / `jax.dlpack.from_dlpack()`. **These capsules alias live GPU memory** — a subsequent `run()` overwrites them in place; clone the tensor if you need a snapshot, and keep the session object alive while the tensor is in use. `ScenarioSweepSession` also *imports* DLPack tensors (`set_injections_dlpack`, `set_gen_v_dlpack`, the `solve_JT_batch_dlpack` inputs): importers validate dtype/device/shape/contiguity (`check_dl_input`), consume the capsule per the protocol, take a `producer_stream` (torch's `current_stream().cuda_stream`) they wait on through an event — session streams are `cudaStreamCreate` (blocking w.r.t. the legacy default stream) but torch user streams are not — and host-sync before returning, since a torch temporary may be recycled by its caching allocator as soon as the call returns. Never remove that sync. ### Differentiable power flow `src/gpusim2grid/differentiable/` wraps `AcPfNrSession` in a `torch.autograd.Function` (`PowerFlowFunction`, `solve_power_flow`). Backward uses the adjoint method (implicit function theorem): it solves Jᵀλ = x̄ via `AcPfNrSession.solve_JT_dlpack` reusing the converged factorization. Sbus is split into real/imag tensors because autograd needs real tensors. Sign conventions and the x-space projection are documented at the top of `_power_flow_op.py` — read it before touching gradients. +**`BatchPowerFlow`** (`differentiable/_batch_pf.py`) is the batched, ML-facing entry point: built from a solved lightsim2grid grid, it owns a `ScenarioSweepGPU` and takes lightsim2grid-`ScenarioSweep`-style per-element inputs — `load_p`, `load_q`, `gen_p`, `gen_v` (all differentiable) and the boolean `line_status`/`trafo_status` masks (**True = connected**, grid2op's convention, unlike lightsim2grid's `set_contingency_lines` where True = tripped). The element→bus assembly is plain torch (`index_add` mirroring `build_bus_injections`), so autograd gives the `load_p`/`load_q`/`gen_p` VJPs for free; an internal `_BatchPowerFlowOp` maps `(P, Q, gen_v) → V` and its backward does the batched adjoint. Its gradient is verified against finite differences in `tests/python/test_batch_power_flow.py` (gradcheck + explicit central differences for every input — keep it that way when touching any of this). The `gen_v` gradient has a direct term (`Re(e^{-jθ}·ḡV)` at the Vm-fixed bus, Python side) and an indirect one (`−λ·∂S_calc/∂Vm_k`, the dS/dVm column `fill_J` never stores for a Vm-fixed bus, computed per slot on the patched Ybus by `gen_v_adjoint_kernel` through a Ybus transpose-position map); only generators whose own bus is Vm-fixed get a non-zero gradient, NaN entries get 0. `torch.autograd.gradcheck` runs many forwards before it back-propagates the first output, so it needs `snapshot_jacobian=True` (the default alias mode refuses that pattern through the `run_counter` guard). + +The C++ pieces this rests on (all `ScenarioSweepSession`, DLPack in/out so a JAX front-end can reuse them): + +- **Driver persistence** — `run()` picks a path against `ScenarioSweepDriverConfig` (the snapshot of n_scenarios/batch_size/strategy/cuDSS config/scaling/mask mode/`fixed_batch_capacity`/base-state generation taken when the driver was built): *cold* (new `BatchPfDriver`: allocation + cuDSS ANALYSIS; `driver_build_counter`), *warm* (only the topology or generator mask changed: a new `ScenarioSweepBatch` built with `forced_batch_size` = the live capacity and swapped in by `BatchPfDriver::replace_source`, no analysis; `source_build_counter`), *hot* (only injections/gen_v changed: one device gather). The policy's `factorized_` flag survives, so reused runs REFACTORIZE only; `mark_reused` zeroes the one-time timing fields so a reused run reports `t_analysis_ms == 0`. Sbus rows live in a canonical ORIGINAL-row-order device buffer (`ScenarioSweepDeviceData::d_Sbus_orig`, filled from numpy or straight from a torch tensor via `set_injections_dlpack`) and are gathered into active-slot order on the device (`gather_rows_kernel`) — the host permutation is gone. `fixed_batch_capacity` uses `batch_size` verbatim as the chunk capacity (no rebalancing over the active count) so the batch stays one chunk whatever rows get islanded. `v_results_dlpack` memory is now overwritten in place across reusing runs (freed only by a cold rebuild). The other two batch sessions still rebuild per `run()`; `replace_source`/`mark_reused` are generic on the driver if they ever want the same. +- **Converged Jacobian** — `keep_final_jacobian` makes `_solve_chunk` refill `d_J_values_batch` at V_final (`nr_fill_J_at_current_V`, the single definition of the fill sequence, shared with the NR loop) after the post-loop residual and before the NaN masking; requires one chunk (the driver throws otherwise). +- **Lazy batched adjoint** — `BatchPfDriver::solve_JT_batch` builds `BatchAdjoint` on its first call only: host counting-sort transpose of the J skeleton + `d_J_to_JT` position map, capacity-sized buffers, a second `CudssBatchSolver` over Jᵀ (one ANALYSIS with the forward's config); every call permutes the values (`transpose_gather_values_kernel`), FACTORIZES once then REFACTORIZES (skipped for a second backward on the same forward), SOLVES, and works in original row order both ways (dropped rows → 0, non-finite rhs → 0). Counters/timings are surfaced as `BatchTimings.adjoint_*` / `t_adjoint_*` (cumulative per driver life, not part of the run() totals). Alias mode reads the chunk buffers (`run_counter` guard on the Python side); snapshot mode passes cloned `j_values_dlpack()` / `ybus_values_dlpack()` / V back in. + ### Types and timing - `dtypes.hpp` defines `real_t` (`float` or `double` per compile flag) and the Eigen/complex aliases. **Use these aliases in new solver code; never hardcode `float`/`double`.** diff --git a/README.md b/README.md index 4c3a867..f0a6c80 100644 --- a/README.md +++ b/README.md @@ -186,6 +186,22 @@ V = sweep.compute(batch_size=512) disconnected = sweep.get_disconnected() # which rows a topology trip islanded ``` +For PyTorch users, the same batch is available as a differentiable layer that +takes lightsim2grid-style per-element inputs (`True = connected` statuses), +solves every row in one GPU pass and back-propagates through it with the +adjoint method — consecutive calls with the same batch size reuse the whole GPU +setup (no new cuDSS analysis, refactorization only), and the transposed system +the backward needs is built lazily on the first `backward()`: + +```python +from gpusim2grid.differentiable import BatchPowerFlow +pf = BatchPowerFlow.from_lsgrid(grid, nb_iter=8) +V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=gen_v, # (n_scen, n_elem) tensors + line_status=line_status, trafo_status=trafo_status) # bool, True = connected +loss = (V.abs() - 1.0).pow(2).sum() +loss.backward() # gradients w.r.t. load_p, load_q, gen_p and gen_v +``` + See [`examples/`](examples/): - `ieee14_basic.py` — end-to-end AC power flow on the IEEE 14-bus case. @@ -196,6 +212,7 @@ See [`examples/`](examples/): - `limit_violations.py` — fused per-contingency bus voltage / branch current limit checking. - `distributed_slack.py` — augmented solve (distributed slack in the Jacobian) via the lightsim2grid bridge. - `differentiable_pf.py` — derivatives through a single power flow via the adjoint method. +- `batch_differentiable_pf.py` — `BatchPowerFlow`: a batch of scenarios (injections, set-points, branch statuses) as one differentiable PyTorch layer, trained over a few steps. ## How it works @@ -253,8 +270,9 @@ design. See the [roadmap](#roadmap) below for areas where help is especially wel Directions we plan to pursue — **any help or ideas are very welcome**: -- **Extend derivatives to the injection sweep**, and later to the full contingency - analysis path (currently differentiation is limited to a single power flow). +- **Extend derivatives** beyond `BatchPowerFlow`'s inputs (`load_p`, `load_q`, + `gen_p`, `gen_v`): generator contingencies (`gen_status`), a JAX front-end on the + same DLPack primitives, and the plain contingency-analysis path. - **Best action selector** — given one or several grid snapshots and a list of candidate actions, find the best action(s) to apply, by evaluating the candidates in batch on the GPU. diff --git a/docs/api.rst b/docs/api.rst index 76a0499..691e9e6 100644 --- a/docs/api.rst +++ b/docs/api.rst @@ -150,6 +150,123 @@ violations") for the equivalent ``ContingencyAnalysisGPU`` walkthroughs — the only difference on :class:`~gpusim2grid.ScenarioSweepGPU` is that each row also carries its own injection. +.. _scenario-sweep-reuse: + +What ``compute()`` reuses across calls +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +The GPU batch driver built by the first ``compute()`` is kept alive by the +sweep and reused by later calls. Each ``compute()`` compares the current +settings with the ones the live driver was built with and takes one of three +paths: + +* **cold** — no driver yet, or something the driver's shape depends on + changed: the number of rows, ``batch_size``, ``strategy``, + ``refactor_period``, a cuDSS choice (``reordering_alg`` / + ``matching_alg`` / ``pivot_epsilon_alg``), the step-scaling knobs, + ``handle_disconnected_grid``, ``fixed_batch_capacity``, or the base state + (``set_contingency_gens`` reserving a different set of buses); +* **warm** — same shape, but ``set_topology`` or ``set_contingency_gens`` was + called since the last ``compute()``; +* **hot** — same shape and topology; only ``set_injections`` / + ``set_injections_from_elements`` / ``set_gen_v`` were called (or nothing). + +``nb_iter`` is applied to the live driver and never forces a rebuild. + +.. list-table:: + :header-rows: 1 + :widths: 34 22 22 22 + + * - Phase + - cold + - warm + - hot + * - Host: per-unit Sbus build from the MW / MVAr matrices (numpy path + only; the device path, ``set_injections_dlpack``, skips it) + - yes, if injections changed + - yes, if injections changed + - yes, if injections changed + * - Host: topology preprocessing — Ybus patch triplets → CSR positions, + connectivity / masking, flat patch arrays, tripped-branch table, + per-row PV pins and slack weights (``ScenarioSweepBatch``) + - yes + - yes + - no + * - Device: allocation of the chunk buffers (V, Ybus values, J values, F, + dx, Ibus), block-diagonal CSR structure, cuSPARSE SpMV descriptor + - yes + - no + - no + * - cuDSS context creation + ANALYSIS (symbolic factorization) + - yes + - no + - no + * - H→D upload of the patch / mask / tripped-branch arrays + - yes + - yes + - no + * - Sbus rows: one H→D copy (numpy) or D→D copy (torch) into the + original-order buffer, then a device gather into batch order + - yes + - yes + - yes + * - ``gen_v`` rows (same copy + gather, Vm-fixed columns only) + - yes, if set + - yes, if set + - only if it changed + * - Per chunk: tile V and Ybus, apply the row's Ybus patches, reseed + ``gen_v``, slice Sbus + - yes + - yes + - yes + * - Newton-Raphson iterations: SpMV, fill F, fill J + - yes + - yes + - yes + * - cuDSS first FACTORIZATION (iteration 0 of the first chunk) + - yes + - no + - no + * - cuDSS REFACTORIZATION (every other iteration, per the strategy) + - yes + - yes + - yes + * - cuDSS SOLVE, voltage update, residuals, result store + - yes + - yes + - yes + * - Limits (``compute_limit_violations``): branch admittance upload / + per-row sentinel reset + - yes / yes + - no / yes + - no / yes + * - Reported one-time timings (``t_analysis_ms``, ``t_alloc_ms``, + ``t_context_init_ms``; ``t_preprocess_ms``, ``t_source_init_ms``) + - measured + - 0 ; measured + - 0 ; 0 + * - Counters bumped + - ``driver_build_counter``, ``source_build_counter`` + - ``source_build_counter`` + - — + +The result buffer behind ``v_results_dlpack()`` is overwritten in place by a +warm or hot call and freed (reallocated) by a cold one — clone the tensor for +a snapshot either way. + +For the differentiable layer (:class:`~gpusim2grid.differentiable.BatchPowerFlow`) +the same table applies to its forward, with one addition when gradients are +requested: the batched Jacobian is refilled at the converged voltages after +the iterations (one extra ``fill J``, on every path). Its backward has its own +lazily built state: the first ``backward()`` transposes the Jacobian pattern, +builds the J→Jᵀ position map and buffers and runs one cuDSS ANALYSIS + one +FACTORIZATION of Jᵀ; every later backward only permutes the values with a +kernel, REFACTORIZES (once per new forward — a second backward on the same +forward only solves) and SOLVES. A cold forward discards that state, and the +next backward rebuilds it. ``timings.adjoint_n_analysis`` / +``adjoint_n_factorize`` / ``adjoint_n_refactorize`` / ``adjoint_n_solve`` +count these over the driver's life. + .. automodule:: gpusim2grid.scenario_sweep :members: :undoc-members: @@ -164,6 +281,40 @@ Single AC power flow Differentiable power flow (alpha) --------------------------------- +Two entry points, both PyTorch ``autograd`` integrations of the GPU solver +using the adjoint method (the converged Jacobian is transposed and factorized +once, then reused for every backward): + +* :class:`~gpusim2grid.differentiable.BatchPowerFlow` -- a **batch** of + scenarios as one differentiable layer, driven by the same per-element inputs + as lightsim2grid's ``ScenarioSweep``: ``load_p``, ``load_q``, ``gen_p``, + ``gen_v`` (all differentiable) and the boolean ``line_status`` / + ``trafo_status`` masks (``True`` = connected). Built once from a solved + lightsim2grid grid; consecutive calls with the same number of rows reuse the + GPU batch driver (no cuDSS analysis, refactorization only), and the + transposed system is built lazily on the first ``backward()`` — see + :ref:`scenario-sweep-reuse` for the phase-by-phase table. +* :func:`~gpusim2grid.differentiable.solve_power_flow` / + :class:`~gpusim2grid.differentiable.PowerFlowFunction` -- a single power flow + from raw ``Sbus`` tensors. + +.. code-block:: python + + from gpusim2grid.differentiable import BatchPowerFlow + + pf = BatchPowerFlow.from_lsgrid(grid, nb_iter=8) # grid: solved lightsim2grid grid + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, # (n_scen, n_load / n_gen) MW, MVAr + gen_v=gen_v, # (n_scen, n_gen) vm_pu, NaN = keep + line_status=line_status, trafo_status=trafo_status) # bool, True = connected + loss = ((V.abs() - 1.0) ** 2).sum() + loss.backward() # d loss / d load_p, ..., d gen_v + +Rows whose branch trips island the grid come back as ``NaN`` (mask them out of +the loss); they get a zero gradient. A forward → forward → backward(first) +pattern is refused with a clear error unless the layer was built with +``snapshot_jacobian=True`` (one extra device copy of the Jacobians per forward), +which is also what ``torch.autograd.gradcheck`` needs. + .. automodule:: gpusim2grid.differentiable :members: :undoc-members: diff --git a/docs/conf.py b/docs/conf.py index c2eed24..806b6aa 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -25,7 +25,7 @@ import gpusim2grid release = gpusim2grid.__version__ except Exception: # pragma: no cover - best effort - release = "0.1.2" + release = "0.2.0.rc0" version = release # -- General configuration --------------------------------------------------- diff --git a/docs/examples.rst b/docs/examples.rst index d3b34d1..47f012c 100644 --- a/docs/examples.rst +++ b/docs/examples.rst @@ -92,3 +92,16 @@ power flow, using the adjoint method via the PyTorch integration. .. literalinclude:: ../examples/differentiable_pf.py :language: python + +Batched differentiable power flow +--------------------------------- + +A whole batch of scenarios (random load scalings and random N-1 line trips) as +one differentiable PyTorch layer (``BatchPowerFlow``): a few optimisation steps +learn a generator redispatch and voltage set-points that flatten the voltage +profile. The reuse counters printed at each step show the GPU batch driver +being built once, the Jacobians being refactorized (never re-analysed) on +later calls, and the transposed system being built on the first backward only. + +.. literalinclude:: ../examples/batch_differentiable_pf.py + :language: python diff --git a/examples/README.md b/examples/README.md index a025d47..df9b6e6 100644 --- a/examples/README.md +++ b/examples/README.md @@ -21,6 +21,7 @@ importable.) | `limit_violations.py` | N-1 screen with `compute_limit_violations=True`: fused, on-device per-contingency bus voltage / branch current / divergence check, reported as a bounded per-contingency violation list. Takes an optional `` argument. | | `distributed_slack.py` | Augmented solve: distributed slack carried in the Jacobian (via the lightsim2grid C++ bridge), matched against the CPU reference. Same path also covers HVDC droop, SVC, and remote voltage control. Needs a bridge-enabled build. | | `differentiable_pf.py` | Derivatives through a single power flow (adjoint method) via the PyTorch autograd integration. Requires PyTorch with CUDA. | +| `batch_differentiable_pf.py` | `BatchPowerFlow`: a whole batch of scenarios (load / generator injections, generator set-points, line / trafo statuses) as one differentiable PyTorch layer; a few optimisation steps, with the GPU-reuse counters printed at each call. Requires PyTorch with CUDA. | `_common.py` is a shared helper (not a standalone example): it wraps lightsim2grid/pandapower to produce the plain NumPy/SciPy arrays the solver diff --git a/pyproject.toml b/pyproject.toml index bdb4ed1..538ee67 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -51,6 +51,11 @@ docs = [ "sphinx-rtd-theme>=2", "myst-parser>=2", ] +# gpusim2grid.differentiable (BatchPowerFlow / PowerFlowFunction) needs a +# CUDA-enabled PyTorch; everything else works without it. +torch = [ + "torch", +] [project.urls] Homepage = "https://github.com/Grid2Op/gpusim2grid" diff --git a/src/_cpp/acpf_nr_kernels.cuh b/src/_cpp/acpf_nr_kernels.cuh index ac0519d..eb7cbd1 100755 --- a/src/_cpp/acpf_nr_kernels.cuh +++ b/src/_cpp/acpf_nr_kernels.cuh @@ -687,4 +687,105 @@ inline void launch_tile(T* dst, const T* src, int n, int batch_size, cudaStream_ tile_kernel<<>>(dst, src, n, batch_size); } +// ============================================================================= +// gather_rows_kernel / launch_gather_rows +// dst[r * n_cols + c] = src[map[r] * n_cols + c] for r in [0, n_rows) +// (map == nullptr → identity: dst row r = src row r). Used to move the +// ORIGINAL-row-order per-scenario data (Sbus rows, adjoint right-hand sides) +// into ACTIVE-slot order on the device, replacing the host permutation the +// batch sources used to do. zero_nonfinite replaces NaN/inf source entries +// by 0 (adjoint rhs of masked/NaN buses). +// scatter_rows_kernel / launch_scatter_rows +// dst[map[r] * n_cols + c] = src[r * n_cols + c] (the inverse move). +// Both have a plain complex / real overload path through the templates below; +// only the real one supports zero_nonfinite. +// ============================================================================= +template +__device__ __forceinline__ T gather_sanitize(T v, bool) { return v; } +template <> +__device__ __forceinline__ float gather_sanitize(float v, bool zero_nonfinite) +{ return (zero_nonfinite && !isfinite(v)) ? 0.f : v; } +template <> +__device__ __forceinline__ double gather_sanitize(double v, bool zero_nonfinite) +{ return (zero_nonfinite && !isfinite(v)) ? 0. : v; } + +template +__global__ void gather_rows_kernel(T* __restrict__ dst, const T* __restrict__ src, + const int* __restrict__ map, + int n_cols, int n_rows, bool zero_nonfinite) +{ + const ptrdiff_t tid = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const ptrdiff_t r = tid / n_cols; + const int c = static_cast(tid % n_cols); + if (r >= n_rows) return; + const ptrdiff_t src_r = map ? map[r] : r; + dst[r * n_cols + c] = gather_sanitize(src[src_r * n_cols + c], zero_nonfinite); +} + +template +inline void launch_gather_rows(T* dst, const T* src, const int* map, + int n_cols, int n_rows, bool zero_nonfinite, cudaStream_t cs) +{ + if (n_cols <= 0 || n_rows <= 0) return; + constexpr int block = 256; + const long long total = static_cast(n_cols) * n_rows; + gather_rows_kernel<<((total + block - 1) / block), block, 0, cs>>>( + dst, src, map, n_cols, n_rows, zero_nonfinite); +} + +template +__global__ void scatter_rows_kernel(T* __restrict__ dst, const T* __restrict__ src, + const int* __restrict__ map, int n_cols, int n_rows) +{ + const ptrdiff_t tid = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const ptrdiff_t r = tid / n_cols; + const int c = static_cast(tid % n_cols); + if (r >= n_rows) return; + const ptrdiff_t dst_r = map ? map[r] : r; + dst[dst_r * n_cols + c] = src[r * n_cols + c]; +} + +template +inline void launch_scatter_rows(T* dst, const T* src, const int* map, + int n_cols, int n_rows, cudaStream_t cs) +{ + if (n_cols <= 0 || n_rows <= 0) return; + constexpr int block = 256; + const long long total = static_cast(n_cols) * n_rows; + scatter_rows_kernel<<((total + block - 1) / block), block, 0, cs>>>( + dst, src, map, n_cols, n_rows); +} + +// ============================================================================= +// gather_cols_rows_kernel / launch_gather_cols_rows +// dst[r * k + j] = src[map[r] * n_cols_src + cols[j]]: row gather (active- +// slot order, map nullable = identity) combined with a column selection -- +// the device path of set_gen_v(), which keeps only the generators whose own +// bus is Vm-fixed (see GenVOverride). +// ============================================================================= +template +__global__ void gather_cols_rows_kernel(T* __restrict__ dst, const T* __restrict__ src, + const int* __restrict__ map, + const int* __restrict__ cols, + int n_cols_src, int k, int n_rows) +{ + const ptrdiff_t tid = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const ptrdiff_t r = tid / k; + const int j = static_cast(tid % k); + if (r >= n_rows) return; + const ptrdiff_t src_r = map ? map[r] : r; + dst[r * k + j] = src[src_r * n_cols_src + cols[j]]; +} + +template +inline void launch_gather_cols_rows(T* dst, const T* src, const int* map, const int* cols, + int n_cols_src, int k, int n_rows, cudaStream_t cs) +{ + if (k <= 0 || n_rows <= 0) return; + constexpr int block = 256; + const long long total = static_cast(k) * n_rows; + gather_cols_rows_kernel<<((total + block - 1) / block), block, 0, cs>>>( + dst, src, map, cols, n_cols_src, k, n_rows); +} + #endif // ACPF_NR_KERNELS_CUH \ No newline at end of file diff --git a/src/_cpp/contingency/batch_pf_driver.cu b/src/_cpp/contingency/batch_pf_driver.cu index eafda18..15fff67 100644 --- a/src/_cpp/contingency/batch_pf_driver.cu +++ b/src/_cpp/contingency/batch_pf_driver.cu @@ -126,6 +126,9 @@ BatchPfDriver::BatchPfDriver( , scaling_max_voltage_change_(scaling_max_voltage_change) , max_dVa_(static_cast(max_dVa)) , max_dVm_(static_cast(max_dVm)) + , reordering_alg_(reordering_alg) + , matching_alg_(matching_alg) + , pivot_epsilon_alg_(pivot_epsilon_alg) { // Pin this driver's stream and allocations to the same device as base. CHK_CUDA_BPF(cudaSetDevice(base.device_id_)); @@ -572,6 +575,13 @@ template BatchTimings BatchPfDriver::solve() { BatchTimings t; + ++n_solves_; + if (keep_final_jacobian_ && n_chunks_ > 1) + throw std::runtime_error( + "[batch_pf] keep_final_jacobian requires the whole batch to be " + "solved in ONE chunk (only the last chunk's Jacobian survives in " + "the chunk buffer): raise batch_size to at least the number of " + "active elements (" + std::to_string(n_active_) + ")."); t.n_contingencies = n_contingencies; t.n_chunks = n_chunks_; t.chunk_size = batch_size_; @@ -805,6 +815,21 @@ void BatchPfDriver::_solve_chunk( } t.t_residual += timer.stop_ms(); + // ------------------------------------------------------------------------- + // ④b Converged Jacobian (opt-in, differentiable wrapper): the loop left + // J(V_{nb_iter-1}) in d_J_values_batch; refill it at V_final using the + // SpMV output of step ④ (d_Ibus_batch = Ybus · V_final for the whole + // padded batch). Same fill sequence as the loop -- one definition + // (nr_fill_J_at_current_V). Must run BEFORE step ⑤'s NaN masking of + // d_V_batch. The forward factors are untouched (cuDSS keeps them in + // its own data object); the batched adjoint reads the values later. + // ------------------------------------------------------------------------- + if (keep_final_jacobian_) { + timer.start(); + nr_fill_J_at_current_V(buf, n_bus, dim_J, nnz_Y, nnz_J, batch_size_, cs); + t.t_fill_J += timer.stop_ms(); + } + // ------------------------------------------------------------------------- // ⑤ Store V results. Without compaction the active slots map contiguously // to result indices (single D→D memcpy). With compaction the original @@ -904,6 +929,303 @@ void BatchPfDriver::_solve_chunk( } } +// ============================================================================= +// Batched adjoint kernels +// ============================================================================= + +// JT_values[b*nnz + perm[i]] = J_values[b*nnz + i] for every slot b: the +// value permutation of the shared J→Jᵀ position map (BatchAdjoint::d_J_to_JT). +__global__ void transpose_gather_values_kernel( + cuda_real_type* __restrict__ dst, + const cuda_real_type* __restrict__ src, + const int* __restrict__ perm, + int nnz, int batch) +{ + const ptrdiff_t tid = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const ptrdiff_t b = tid / nnz; + const int i = static_cast(tid % nnz); + if (b >= batch) return; + dst[b * nnz + perm[i]] = src[b * nnz + i]; +} + +// gen_v adjoint: for every (slot, Vm-fixed bus k) the contraction of the +// adjoint vector with the dS_calc/dVm_k column (which J never stores for a +// Vm-fixed bus): +// λ̃_i = λ[p_row(i)] + j λ[q_row(i)] (0 where the bus owns no row) +// w_i = conj(λ̃_i) V_i +// g_k = − Re( e^{−jθ_k} · Σ_{i∈col k} conj(Y_ik) w_i + conj(λ̃_k) e^{jθ_k} conj(I_k) ) +// Y_ik is read through the Ybus transpose-position map (row k's pattern is +// column k's pattern: the pattern is symmetric); I_k = Σ_j Y_kj V_j is +// recomputed inline from row k. Non-finite V (masked buses) count as 0; a +// non-finite V_k gives g_k = 0. The minus sign is the implicit-function +// sign: dx/dVm_k = −J⁻¹ c_k (see _batch_pf.py's docstring). +__global__ void gen_v_adjoint_kernel( + cuda_real_type* __restrict__ d_gvm, // [batch × n_bus] out (slot order) + const cuda_real_type* __restrict__ d_lam, // [batch × dim_J] (slot order) + const cudaComplexType* __restrict__ d_V, // [batch × n_bus] (slot order) + const cudaComplexType* __restrict__ d_Yvals, // [batch × nnz_Y] (slot order) + const int* __restrict__ d_Y_outer, + const int* __restrict__ d_Y_inner, + const int* __restrict__ d_YT_pos, + const int* __restrict__ d_p_row_of_bus, + const int* __restrict__ d_q_row_of_bus, + const char* __restrict__ d_is_vm_fixed, + int n_bus, int nnz_Y, int dim_J, int batch) +{ + const ptrdiff_t tid = static_cast(blockIdx.x) * blockDim.x + threadIdx.x; + const ptrdiff_t b = tid / n_bus; + const int k = static_cast(tid % n_bus); + if (b >= batch) return; + + cuda_real_type g = 0; + if (d_is_vm_fixed[k]) { + const cudaComplexType* V = d_V + b * n_bus; + const cudaComplexType* Yv = d_Yvals + b * nnz_Y; + const cuda_real_type* lam = d_lam + b * dim_J; + + auto lam_tilde = [&](int i) -> cudaComplexType { + const int pr = d_p_row_of_bus[i], qr = d_q_row_of_bus[i]; + return CudaFunHelper::my_make_cuComplex( + pr >= 0 ? lam[pr] : cuda_real_type(0), + qr >= 0 ? lam[qr] : cuda_real_type(0)); + }; + auto finite_or_zero = [](cudaComplexType v) -> cudaComplexType { + if (!isfinite(v.x) || !isfinite(v.y)) + return CudaFunHelper::my_make_cuComplex(cuda_real_type(0), cuda_real_type(0)); + return v; + }; + + const cudaComplexType Vk = V[k]; + const cuda_real_type mag = CudaFunHelper::my_cuCabs(Vk); + if (isfinite(Vk.x) && isfinite(Vk.y) && mag > cuda_real_type(0)) { + const cudaComplexType e = CudaFunHelper::my_make_cuComplex(Vk.x / mag, Vk.y / mag); + cudaComplexType Ik = CudaFunHelper::my_make_cuComplex(cuda_real_type(0), cuda_real_type(0)); + cudaComplexType acc = Ik; + for (int p = d_Y_outer[k]; p < d_Y_outer[k + 1]; ++p) { + const int j = d_Y_inner[p]; + const cudaComplexType Vj = finite_or_zero(V[j]); + Ik = CudaFunHelper::my_cuCadd(Ik, CudaFunHelper::my_cuCmul(Yv[p], Vj)); + const int pT = d_YT_pos[p]; + if (pT >= 0) { + const cudaComplexType Yjk = Yv[pT]; + const cudaComplexType wj = CudaFunHelper::my_cuCmul( + CudaFunHelper::my_cuConj(lam_tilde(j)), Vj); + acc = CudaFunHelper::my_cuCadd(acc, + CudaFunHelper::my_cuCmul(CudaFunHelper::my_cuConj(Yjk), wj)); + } + } + const cudaComplexType t1 = CudaFunHelper::my_cuCmul(CudaFunHelper::my_cuConj(e), acc); + const cudaComplexType t2 = CudaFunHelper::my_cuCmul( + CudaFunHelper::my_cuCmul(CudaFunHelper::my_cuConj(lam_tilde(k)), e), + CudaFunHelper::my_cuConj(Ik)); + g = -(t1.x + t2.x); + } + } + d_gvm[b * n_bus + k] = g; +} + +// ============================================================================= +// _prepare_adjoint — first solve_JT_batch() call only. +// ============================================================================= +template +void BatchPfDriver::_prepare_adjoint() +{ + auto t_build_start = std::chrono::steady_clock::now(); + adjoint_ = std::make_unique(); + BatchAdjoint& A = *adjoint_; + + const int dim_J = base.dim_J; + const int nnz_J = base.nnz_J; + + // Host transpose of the shared J skeleton: counting sort over columns. + thrust::host_vector h_outer(base.d_J_outer); + thrust::host_vector h_inner(base.d_J_inner); + std::vector h_JT_outer(static_cast(dim_J) + 1, 0); + std::vector h_JT_inner(static_cast(nnz_J), 0); + std::vector h_map(static_cast(nnz_J), 0); + for (int p = 0; p < nnz_J; ++p) ++h_JT_outer[static_cast(h_inner[p]) + 1]; + for (int j = 0; j < dim_J; ++j) h_JT_outer[static_cast(j) + 1] += h_JT_outer[static_cast(j)]; + std::vector fill(h_JT_outer.begin(), h_JT_outer.end() - 1); + for (int i = 0; i < dim_J; ++i) + for (int p = h_outer[i]; p < h_outer[i + 1]; ++p) { + const int j = h_inner[p]; + const int pos = fill[static_cast(j)]++; + h_JT_inner[static_cast(pos)] = i; // rows come out sorted + h_map[static_cast(p)] = pos; + } + upload_h2d(A.d_JT_outer, h_JT_outer.data(), h_JT_outer.size(), cs); + upload_h2d(A.d_JT_inner, h_JT_inner.data(), h_JT_inner.size(), cs); + upload_h2d(A.d_J_to_JT, h_map.data(), h_map.size(), cs); + + A.d_JT_values.resize(static_cast(batch_size_) * nnz_J); + A.d_rhs.resize(static_cast(batch_size_) * dim_J); + A.d_sol.resize(static_cast(batch_size_) * dim_J); + A.d_sol_full.resize(static_cast(n_contingencies) * dim_J); + zero_d(A.d_rhs, cs); + zero_d(A.d_sol, cs); + zero_d(A.d_sol_full, cs); + + // Second uniform-batch cuDSS context over the transposed skeleton, same + // config as the forward (so a reordering workaround applies to both). + A.solver.initialize( + batch_size_, dim_J, nnz_J, + thrust::raw_pointer_cast(A.d_JT_outer.data()), + thrust::raw_pointer_cast(A.d_JT_inner.data()), + thrust::raw_pointer_cast(A.d_JT_values.data()), + thrust::raw_pointer_cast(A.d_rhs.data()), + thrust::raw_pointer_cast(A.d_sol.data()), + cs, reordering_alg_, matching_alg_, pivot_epsilon_alg_); + A.n_analysis = 1; + cs.synchronize(); + A.t_build_ms = bpf_ms_since(t_build_start); +} + +template +void BatchPfDriver::_prepare_gen_v_adjoint(const std::vector& is_vm_fixed_bus) +{ + BatchAdjoint& A = *adjoint_; + const int n_bus = base.n_bus; + const int nnz_Y = base.nnz_Y; + if (static_cast(is_vm_fixed_bus.size()) != n_bus) + throw std::runtime_error( + "[batch_pf] solve_JT_batch: is_vm_fixed_bus must have n_bus entries"); + + // Ybus transpose-position map: for entry p = (i, j), the position of + // (j, i) in the same CSR (−1 if the pattern is not symmetric there). + thrust::host_vector h_outer(base.d_Ybus_outer); + thrust::host_vector h_inner(base.d_Ybus_inner); + std::vector h_pos(static_cast(nnz_Y), -1); + for (int i = 0; i < n_bus; ++i) + for (int p = h_outer[i]; p < h_outer[i + 1]; ++p) { + const int j = h_inner[p]; + // binary search for column i in row j (rows are column-sorted) + int lo = h_outer[j], hi = h_outer[j + 1] - 1; + while (lo <= hi) { + const int mid = (lo + hi) / 2; + const int c = h_inner[mid]; + if (c == i) { h_pos[static_cast(p)] = mid; break; } + else if (c < i) lo = mid + 1; + else hi = mid - 1; + } + } + upload_h2d(A.d_Ybus_T_pos, h_pos.data(), h_pos.size(), cs); + upload_h2d(A.d_p_row_of_bus, base.h_p_row_of_bus.data(), base.h_p_row_of_bus.size(), cs); + upload_h2d(A.d_q_row_of_bus, base.h_q_row_of_bus.data(), base.h_q_row_of_bus.size(), cs); + upload_h2d(A.d_is_vm_fixed_bus, is_vm_fixed_bus.data(), is_vm_fixed_bus.size(), cs); + A.d_gvm.resize(static_cast(batch_size_) * n_bus); + A.d_gvm_full.resize(static_cast(n_contingencies) * n_bus); + zero_d(A.d_gvm, cs); + zero_d(A.d_gvm_full, cs); + cs.synchronize(); + A.gen_v_ready = true; +} + +// ============================================================================= +// solve_JT_batch +// ============================================================================= +template +void BatchPfDriver::solve_JT_batch( + const cuda_real_type* d_rhs_orig, + const cuda_real_type* d_J_ext, + bool want_gen_v_grad, + const cudaComplexType* d_Ybus_ext, + const cudaComplexType* d_V_ext_orig, + const std::vector& is_vm_fixed_bus) +{ + if (n_solves_ == 0) + throw std::runtime_error( + "[batch_pf] solve_JT_batch: call solve() (a forward run) first"); + if (!adjoint_) _prepare_adjoint(); + BatchAdjoint& A = *adjoint_; + + const int dim_J = base.dim_J; + const int nnz_J = base.nnz_J; + const int n_bus = base.n_bus; + const int nnz_Y = base.nnz_Y; + const int* d_map = source_.d_result_map(); // nullptr → identity + CudaTimer timer(cs); + + // ① Permute J → Jᵀ values and (re)factorize, unless this driver's own J + // was already permuted + factorized since its last solve() (a second + // backward on the same forward, e.g. retain_graph=True). + const bool own_J = (d_J_ext == nullptr); + const bool need_refactor = !A.factorized || !own_J || A.factorized_for_solve != n_solves_; + if (need_refactor) { + const cuda_real_type* src = own_J ? thrust::raw_pointer_cast(d_J_values_batch.data()) : d_J_ext; + transpose_gather_values_kernel<<< + nr_grid_size((long long)batch_size_ * nnz_J, BS), BS, 0, cs>>>( + thrust::raw_pointer_cast(A.d_JT_values.data()), src, + thrust::raw_pointer_cast(A.d_J_to_JT.data()), nnz_J, batch_size_); + CHK_CUDA_BPF(cudaGetLastError()); + A.solver.set_values(thrust::raw_pointer_cast(A.d_JT_values.data())); + timer.start(); + if (!A.factorized) { + A.solver.factor(); + A.t_first_factorize += timer.stop_ms(); + ++A.n_factorize; + A.factorized = true; + } else { + A.solver.refactor(); + A.t_refactorize += timer.stop_ms(); + ++A.n_refactorize; + } + A.factorized_for_solve = own_J ? n_solves_ : -1; + } + + // ② Right-hand side: original order → slot order (non-finite → 0), phantom + // slots zero (their base-case Jᵀ then yields λ = 0). + zero_d(A.d_rhs, cs); + launch_gather_rows(thrust::raw_pointer_cast(A.d_rhs.data()), d_rhs_orig, d_map, + dim_J, n_active_, /*zero_nonfinite=*/true, cs); + CHK_CUDA_BPF(cudaGetLastError()); + + // ③ Solve and scatter back to original order. + timer.start(); + A.solver.solve(); + A.t_solve += timer.stop_ms(); + ++A.n_solve; + zero_d(A.d_sol_full, cs); + launch_scatter_rows(thrust::raw_pointer_cast(A.d_sol_full.data()), + thrust::raw_pointer_cast(A.d_sol.data()), d_map, + dim_J, n_active_, cs); + CHK_CUDA_BPF(cudaGetLastError()); + + // ④ gen_v contraction (optional). + if (want_gen_v_grad) { + if (!A.gen_v_ready) _prepare_gen_v_adjoint(is_vm_fixed_bus); + const cudaComplexType* d_V_slots; + if (d_V_ext_orig) { + A.d_V_ext_slots.resize(static_cast(batch_size_) * n_bus); + launch_gather_rows(thrust::raw_pointer_cast(A.d_V_ext_slots.data()), d_V_ext_orig, + d_map, n_bus, n_active_, /*zero_nonfinite=*/false, cs); + d_V_slots = thrust::raw_pointer_cast(A.d_V_ext_slots.data()); + } else { + d_V_slots = thrust::raw_pointer_cast(d_V_batch.data()); + } + const cudaComplexType* d_Y_slots = + d_Ybus_ext ? d_Ybus_ext : thrust::raw_pointer_cast(d_Ybus_values_batch.data()); + gen_v_adjoint_kernel<<>>( + thrust::raw_pointer_cast(A.d_gvm.data()), + thrust::raw_pointer_cast(A.d_sol.data()), + d_V_slots, d_Y_slots, + thrust::raw_pointer_cast(base.d_Ybus_outer.data()), + thrust::raw_pointer_cast(base.d_Ybus_inner.data()), + thrust::raw_pointer_cast(A.d_Ybus_T_pos.data()), + thrust::raw_pointer_cast(A.d_p_row_of_bus.data()), + thrust::raw_pointer_cast(A.d_q_row_of_bus.data()), + thrust::raw_pointer_cast(A.d_is_vm_fixed_bus.data()), + n_bus, nnz_Y, dim_J, n_active_); + CHK_CUDA_BPF(cudaGetLastError()); + zero_d(A.d_gvm_full, cs); + launch_scatter_rows(thrust::raw_pointer_cast(A.d_gvm_full.data()), + thrust::raw_pointer_cast(A.d_gvm.data()), d_map, + n_bus, n_active_, cs); + CHK_CUDA_BPF(cudaGetLastError()); + } + + cs.synchronize(); +} + // ============================================================================= // Explicit template instantiations // ============================================================================= diff --git a/src/_cpp/contingency/batch_pf_driver.cuh b/src/_cpp/contingency/batch_pf_driver.cuh index 3751d7b..febb948 100644 --- a/src/_cpp/contingency/batch_pf_driver.cuh +++ b/src/_cpp/contingency/batch_pf_driver.cuh @@ -56,7 +56,12 @@ #include "strategies/policy_iter0_only.cuh" #include "strategies/policy_refactor_every_n.cuh" +#include +#include +#include +#include #include +#include #include "Eigen/Core" #include "Eigen/SparseCore" @@ -75,6 +80,52 @@ struct BatchPfDriverContext { int nnz_Y; }; +// ----------------------------------------------------------------------------- +// BatchAdjoint — the batched transposed system Jᵀ λ = x̄ (one per batch slot), +// built LAZILY by BatchPfDriver::solve_JT_batch on its first call and reused +// by every later call (requirements: nothing exists until the first backward; +// later backward calls only permute values, REFACTORIZE and SOLVE). +// +// cuDSS (0.8) has no transposed-solve mode, so Jᵀ is an explicit second +// uniform-batch system over the SAME capacity: its CSR skeleton is the +// transpose of the shared J skeleton (host counting sort, once), and its +// values are J's values permuted through d_J_to_JT (JT_values[map[i]] = +// J_values[i]) by one kernel per call. Everything is sized by the driver's +// fixed capacity (batch_size_), so it survives replace_source() and dies with +// the driver. +// ----------------------------------------------------------------------------- +struct BatchAdjoint { + CudssBatchSolver solver; // own cuDSS context + ANALYSIS + + thrust::device_vector d_JT_outer; // [dim_J + 1] + thrust::device_vector d_JT_inner; // [nnz_J] + thrust::device_vector d_J_to_JT; // [nnz_J] position map + thrust::device_vector d_JT_values; // [capacity × nnz_J] + thrust::device_vector d_rhs; // [capacity × dim_J], slot order + thrust::device_vector d_sol; // [capacity × dim_J], slot order + thrust::device_vector d_sol_full; // [n_contingencies × dim_J], original order + + // gen_v adjoint (dS/dVm column contraction at Vm-fixed buses), built on + // the first call that asks for it. + bool gen_v_ready = false; + thrust::device_vector d_Ybus_T_pos; // [nnz_Y] position of (j,i) for entry (i,j) + thrust::device_vector d_p_row_of_bus; // [n_bus] + thrust::device_vector d_q_row_of_bus; // [n_bus] + thrust::device_vector d_is_vm_fixed_bus; // [n_bus] + thrust::device_vector d_V_ext_slots; // [capacity × n_bus] snapshot-mode scratch + thrust::device_vector d_gvm; // [capacity × n_bus], slot order + thrust::device_vector d_gvm_full; // [n_contingencies × n_bus], original order + + bool factorized = false; + int factorized_for_solve = -1; // driver n_solves_ whose own J was last permuted+refactored + + // Counters / timings (cumulative over the driver's life; surfaced through + // BatchTimings by the sessions). + int n_analysis = 0, n_factorize = 0, n_refactorize = 0, n_solve = 0; + double t_build_ms = 0.; + TimingEntry t_first_factorize, t_refactorize, t_solve; +}; + // ============================================================================= // BatchPfDriver // ============================================================================= @@ -248,6 +299,29 @@ struct BatchPfDriver { PolicyIter0Only, PolicyRefactorEveryN> policy_; + // cuDSS analysis config, retained so the lazily built adjoint context + // (BatchAdjoint) analyses Jᵀ with the same choices as the forward. + ReorderingAlg reordering_alg_ = ReorderingAlg::Default; + MatchingAlg matching_alg_ = MatchingAlg::None; + PivotEpsilonAlg pivot_epsilon_alg_ = PivotEpsilonAlg::Default; + + // ------------------------------------------------------------------------- + // Persistence / adjoint state + // + // keep_final_jacobian_ : after the NR loop of a chunk, refill + // d_J_values_batch at the CONVERGED V (the loop + // leaves J(V_{nb_iter-1}) behind). Only one chunk + // may be solved then (only the last chunk's J + // survives in the chunk buffer) -- solve() throws + // otherwise. Set by the differentiable wrapper. + // n_solves_ : solve() calls on this driver; the adjoint keys + // its "J already permuted + refactored" cache on it. + // adjoint_ : null until the first solve_JT_batch(). + // ------------------------------------------------------------------------- + bool keep_final_jacobian_ = false; + int n_solves_ = 0; + std::unique_ptr adjoint_; + // ------------------------------------------------------------------------- // Constructor // ------------------------------------------------------------------------- @@ -287,6 +361,107 @@ struct BatchPfDriver { // ------------------------------------------------------------------------- BatchTimings solve(); + // ------------------------------------------------------------------------- + // replace_source — swap in a NEW BatchSource on a LIVE driver (the "warm" + // path of ScenarioSweepSession::run(): new topology, same shape). Keeps + // the cuDSS context + ANALYSIS, the chunk buffers, the SpMV descriptor and + // the policy state (a policy that already factorized refactorizes next), + // re-running only the source's own H→D setup. The new source must have + // been built for this driver's capacity (used_batch_size() == batch_size_) + // so its chunk ranges line up with the chunk loop; its active count may + // shrink or grow within n_contingencies (phantom padding handles a short + // last chunk). A member template so the explicit class instantiations of + // sources that are not move-assignable do not instantiate it. + // ------------------------------------------------------------------------- + template + void replace_source(S&& src) + { + cs.synchronize(); + if (src.used_batch_size() != batch_size_) + throw std::runtime_error( + "[batch_pf] replace_source: the new source was built for a chunk " + "size of " + std::to_string(src.used_batch_size()) + " but this " + "driver's capacity is " + std::to_string(batch_size_)); + if (src.n_active() > n_contingencies) + throw std::runtime_error( + "[batch_pf] replace_source: more active elements than result slots"); + source_ = std::move(src); + n_active_ = source_.n_active(); + n_chunks_ = (n_active_ + batch_size_ - 1) / batch_size_; + t_preprocess_ms_ = source_.cpu_preprocess_ms(); + + auto t_source_start = std::chrono::steady_clock::now(); + { + BatchPfDriverContext ctx = make_context(); + source_.initialize(ctx, cs); + } + cs.synchronize(); + t_source_init_ms_ = std::chrono::duration( + std::chrono::steady_clock::now() - t_source_start).count(); + if (adjoint_) adjoint_->factorized_for_solve = -1; + } + + // ------------------------------------------------------------------------- + // mark_reused — a session reusing this driver for another run() calls + // this first so the one-time construction costs (allocation, cuDSS + // ANALYSIS, context creation) are reported once, not on every run; on the + // "hot" path (no new source at all) the source's own preprocess/upload + // costs are zeroed too. The absolute numbers are what let a caller (or a + // test) tell "reused" from "rebuilt". + // ------------------------------------------------------------------------- + void mark_reused(bool hot) + { + t_alloc_ms_ = 0.; + t_analysis_ms_ = 0.; + t_context_init_ms_ = 0.; + if (hot) { + t_preprocess_ms_ = 0.; + t_source_init_ms_ = 0.; + } + } + + // ------------------------------------------------------------------------- + // solve_JT_batch — batched adjoint solve Jᵀ λ = x̄ per active slot, in + // ORIGINAL row order both ways (see BatchAdjoint above; first call builds + // everything). Requires solve() to have run with keep_final_jacobian_ so + // d_J_values_batch holds the converged J (alias mode), or a caller-owned + // [capacity × nnz_J] snapshot of those values (d_J_ext, snapshot mode). + // + // d_rhs_orig : [n_contingencies × dim_J]; non-finite entries → 0. + // d_J_ext : nullptr → use d_J_values_batch. + // want_gen_v_grad : also compute the gen_v (Vm-fixed bus) gradient + // contraction into d_gvm_full (see gen_v_adjoint_kernel). + // d_Ybus_ext : snapshot of [capacity × nnz_Y] patched Ybus values + // (nullptr → d_Ybus_values_batch). gen_v only. + // d_V_ext_orig : [n_contingencies × n_bus] converged V in original + // order (nullptr → the chunk's own d_V_batch). gen_v only. + // is_vm_fixed_bus : [n_bus] (pv ∪ slack) mask, gen_v only. + // Results: d_JT_sol_full_ptr() [n_contingencies × dim_J] and, when asked, + // d_gvm_full_ptr() [n_contingencies × n_bus]; rows of compacted-out + // (islanded) elements are 0. Synchronizes cs before returning. + // ------------------------------------------------------------------------- + void solve_JT_batch(const cuda_real_type* d_rhs_orig, + const cuda_real_type* d_J_ext, + bool want_gen_v_grad, + const cudaComplexType* d_Ybus_ext, + const cudaComplexType* d_V_ext_orig, + const std::vector& is_vm_fixed_bus); + + bool adjoint_ready() const { return static_cast(adjoint_); } + const cuda_real_type* d_JT_sol_full_ptr() const { + return adjoint_ ? thrust::raw_pointer_cast(adjoint_->d_sol_full.data()) : nullptr; + } + const cuda_real_type* d_gvm_full_ptr() const { + return (adjoint_ && !adjoint_->d_gvm_full.empty()) + ? thrust::raw_pointer_cast(adjoint_->d_gvm_full.data()) : nullptr; + } + const cuda_real_type* j_values_ptr() const { + return thrust::raw_pointer_cast(d_J_values_batch.data()); + } + const cudaComplexType* ybus_values_ptr() const { + return thrust::raw_pointer_cast(d_Ybus_values_batch.data()); + } + // ------------------------------------------------------------------------- // copy_results_to_host — syncs cs and copies V_results + residuals. // ------------------------------------------------------------------------- @@ -375,6 +550,12 @@ struct BatchPfDriver { private: void _solve_chunk(int c_start, int actual_batch, BatchTimings& t); + // First-call setup of the adjoint: skeleton transpose + position map, + // buffers, cuDSS ANALYSIS of Jᵀ (with the forward's config). + void _prepare_adjoint(); + // First-call setup of the gen_v contraction data (Ybus transpose-position + // map, bus→row maps, Vm-fixed mask). + void _prepare_gen_v_adjoint(const std::vector& is_vm_fixed_bus); }; #endif // BATCH_PF_DRIVER_CUH \ No newline at end of file diff --git a/src/_cpp/contingency/batch_sources/scenario_sweep_batch.cu b/src/_cpp/contingency/batch_sources/scenario_sweep_batch.cu index 703dccb..80cf386 100644 --- a/src/_cpp/contingency/batch_sources/scenario_sweep_batch.cu +++ b/src/_cpp/contingency/batch_sources/scenario_sweep_batch.cu @@ -23,10 +23,11 @@ inline void _chk_cuda(cudaError_t e, const char* what) { } // ============================================================================= -// initialize — upload flat Ybus-patch arrays + active-permuted Sbus rows; -// allocate the per-chunk Sbus buffer. Mirrors ContingencyBatch::initialize -// (patch upload) + InjectionBatch::initialize (Sbus upload / buffer alloc), -// minus InjectionBatch's one-time Ybus tiling (Ybus varies per chunk here). +// initialize — upload flat Ybus-patch arrays, masks, tripped table, active +// map; size the Sbus buffers. Mirrors ContingencyBatch::initialize (patch +// upload) + InjectionBatch::initialize (buffer alloc), minus InjectionBatch's +// one-time Ybus tiling (Ybus varies per chunk here) and minus the Sbus upload +// (set_sbus_from_orig gathers it from the session's device buffer). // ============================================================================= void ScenarioSweepBatch::initialize(BatchPfDriverContext& ctx, cudaStream_t cs) { @@ -34,10 +35,12 @@ void ScenarioSweepBatch::initialize(BatchPfDriverContext& ctx, cudaStream_t cs) upload_h2d(d_flat_k, h_flat_k_.data(), h_flat_k_.size(), cs); upload_h2d(d_flat_delta_re, h_flat_delta_re_.data(), h_flat_delta_re_.size(), cs); upload_h2d(d_flat_delta_im, h_flat_delta_im_.data(), h_flat_delta_im_.size(), cs); - if (!active_to_orig_.empty() - && static_cast(active_to_orig_.size()) < n_total_) + // Always uploaded (identity included): the row gathers index through it. + if (!active_to_orig_.empty()) upload_h2d(d_active_to_orig, active_to_orig_.data(), active_to_orig_.size(), cs); + else + d_active_to_orig.clear(); // handle_disconnected_grid masking / PV-pin / stranded-controller entries // (only when any exist). @@ -55,16 +58,14 @@ void ScenarioSweepBatch::initialize(BatchPfDriverContext& ctx, cudaStream_t cs) h_trip_branch_flat_.size(), cs); } - if (static_cast(h_Sbus_all_.size()) - != n_active() * ctx.n_bus) { + if (n_bus_ != ctx.n_bus) throw std::runtime_error( - "[scenario_sweep_batch] h_Sbus_all_ size does not match n_active * n_bus"); - } - upload_h2d(d_Sbus_all, h_Sbus_all_.data(), h_Sbus_all_.size(), cs); + "[scenario_sweep_batch] bus count does not match the driver's"); + d_Sbus_all.resize(static_cast(n_active()) * ctx.n_bus); d_Sbus_batch.resize(static_cast(ctx.batch_size) * ctx.n_bus); - // set_gen_v() overrides, if any -- see GenVOverride's own doc. - if (gen_v_override_.k_active() > 0) { + // set_gen_v() overrides given to the ctor (host path), if any. + if (gen_v_override_.k_active() > 0 && !gen_v_override_.h_gen_v_all.empty()) { upload_h2d(d_gv_active_bus, gen_v_override_.h_active_bus.data(), gen_v_override_.h_active_bus.size(), cs); upload_h2d(d_gv_all, gen_v_override_.h_gen_v_all.data(), @@ -82,6 +83,57 @@ void ScenarioSweepBatch::initialize(BatchPfDriverContext& ctx, cudaStream_t cs) } } +// ============================================================================= +// set_sbus_from_orig — one gather kernel, original row order → active slots. +// ============================================================================= +void ScenarioSweepBatch::set_sbus_from_orig(const cudaComplexType* d_Sbus_orig, + cudaStream_t cs) +{ + const int n_act = n_active(); + if (n_act <= 0) return; + if (d_Sbus_all.size() != static_cast(n_act) * n_bus_) + throw std::runtime_error( + "[scenario_sweep_batch] set_sbus_from_orig: call initialize() first"); + launch_gather_rows(thrust::raw_pointer_cast(d_Sbus_all.data()), d_Sbus_orig, + d_active_to_orig_ptr(), n_bus_, n_act, + /*zero_nonfinite=*/false, cs); + _chk_cuda(cudaGetLastError(), "Sbus row gather"); +} + +// ============================================================================= +// set_gen_v (host path) / set_gen_v_from_orig (device path) +// ============================================================================= +void ScenarioSweepBatch::set_gen_v(GenVOverride&& gen_v_override_orig, cudaStream_t cs) +{ + gen_v_override_ = GenVOverride{}; + if (gen_v_override_orig.k_active() <= 0) return; + _permute_gen_v_rows(std::move(gen_v_override_orig)); + upload_h2d(d_gv_active_bus, gen_v_override_.h_active_bus.data(), + gen_v_override_.h_active_bus.size(), cs); + upload_h2d(d_gv_all, gen_v_override_.h_gen_v_all.data(), + gen_v_override_.h_gen_v_all.size(), cs); +} + +void ScenarioSweepBatch::set_gen_v_from_orig(const cuda_real_type* d_gen_v_orig, int n_gen, + const std::vector& active_cols, + const std::vector& active_bus, + cudaStream_t cs) +{ + gen_v_override_ = GenVOverride{}; + const int k = static_cast(active_cols.size()); + if (k <= 0 || active_bus.size() != active_cols.size()) return; + gen_v_override_.h_active_bus = active_bus; // k_active() > 0 gates the reseed + upload_h2d(d_gv_active_bus, active_bus.data(), active_bus.size(), cs); + upload_h2d(d_gv_active_col, active_cols.data(), active_cols.size(), cs); + const int n_act = n_active(); + d_gv_all.resize(static_cast(n_act) * k); + launch_gather_cols_rows(thrust::raw_pointer_cast(d_gv_all.data()), d_gen_v_orig, + d_active_to_orig_ptr(), + thrust::raw_pointer_cast(d_gv_active_col.data()), + n_gen, k, n_act, cs); + _chk_cuda(cudaGetLastError(), "gen_v gather"); +} + // ============================================================================= // prepare_Ybus_batch — verbatim ContingencyBatch::prepare_Ybus_batch. // ============================================================================= diff --git a/src/_cpp/contingency/batch_sources/scenario_sweep_batch.cuh b/src/_cpp/contingency/batch_sources/scenario_sweep_batch.cuh index cf2bedc..17ca094 100644 --- a/src/_cpp/contingency/batch_sources/scenario_sweep_batch.cuh +++ b/src/_cpp/contingency/batch_sources/scenario_sweep_batch.cuh @@ -25,10 +25,15 @@ // skipping when the angle reference or a controller bus is stranded). // - Sbus side: identical to InjectionBatch's dense per-scenario (n_scenario // × n_bus) complex Sbus, sbus_stride = n_bus. The one wrinkle: since -// compaction can drop rows, h_Sbus_all_ is built ALREADY PERMUTED into -// active-slot order at construction (h_Sbus_all_[slot] = original row -// active_to_orig_[slot]) — prepare_Sbus_batch's row-slice copy is then -// verbatim InjectionBatch code, needing no extra indirection. +// compaction can drop rows, d_Sbus_all holds the rows in ACTIVE-slot +// order (d_Sbus_all[slot] = original row active_to_orig_[slot]) — +// prepare_Sbus_batch's row-slice copy is then verbatim InjectionBatch +// code. The rows are NOT permuted on the host any more: the session owns +// a canonical ORIGINAL-row-order device buffer (filled either from numpy +// or straight from a torch tensor) and set_sbus_from_orig() gathers it +// into active-slot order with one kernel (gather_rows_kernel), on every +// run() path — cold (new driver), warm (new source on a live driver) or +// hot (new injections only). See ScenarioSweepSession::run(). // // Per-chunk behaviour // ------------------- @@ -44,6 +49,14 @@ // (check_limit_violations_kernel, batch_pf_driver.cu) are already generic // over any BatchSource, so no changes are needed outside this file/session to // support either handle_disconnected_grid or compute_limit_violations here. +// +// Driver persistence +// ------------------ +// A source can be built for an ALREADY LIVE BatchPfDriver (warm path): pass +// forced_batch_size = the driver's capacity so the chunk ranges built here +// match the driver's chunk loop (the driver refuses a mismatch), then +// BatchPfDriver::replace_source() move-assigns it in — hence the defaulted +// move assignment below. // ============================================================================= #include @@ -55,7 +68,7 @@ #include "../../cuda_utils.h" #include "../../cu_complex_utils.h" #include "../../timing_utils.hpp" -#include "../../acpf_nr_kernels.cuh" // apply_contingencies_kernel, apply_gen_v_kernel +#include "../../acpf_nr_kernels.cuh" // apply_contingencies_kernel, apply_gen_v_kernel, gather kernels #include "../../acpf_nr_state.cuh" #include "../../contingency_analysis_helper.hpp" #include "../../nr_iter_step.cuh" // BS @@ -88,11 +101,17 @@ struct ScenarioSweepBatch { std::vector chunk_ranges_; int n_total_ = 0; + int n_bus_ = 0; std::vector active_to_orig_; + // ALWAYS uploaded (even without compaction, where it is the identity): + // the Sbus / gen_v / adjoint row gathers index through it. d_result_map() + // still returns nullptr when identity so the driver keeps its contiguous + // result-store fast path. thrust::device_vector d_active_to_orig; // Effective per-chunk size, rebalanced over the ACTIVE (simulated) count - // — read back by the session and handed to the driver. + // — read back by the session and handed to the driver — or forced to the + // live driver's capacity (warm path). int used_batch_size_ = 0; thrust::device_vector d_flat_ctg_id; @@ -115,9 +134,9 @@ struct ScenarioSweepBatch { // ------------------------------------------------------------------------- // Per-row distributed-slack weights (generator contingencies, see // ScenarioSweepSession::set_contingency_gens). h_slack_w_all_ is - // [n_active * n_slack], ALREADY PERMUTED into active-slot order like - // h_Sbus_all_; empty when no row re-weights the slack (every slot then - // keeps base's shared weights, slack_w_stride 0 -- bit-identical). + // [n_active * n_slack], ALREADY PERMUTED into active-slot order; empty when + // no row re-weights the slack (every slot then keeps base's shared + // weights, slack_w_stride 0 -- bit-identical). // ------------------------------------------------------------------------- std::vector h_slack_w_all_; int n_slack_ = 0; @@ -135,77 +154,75 @@ struct ScenarioSweepBatch { thrust::device_vector d_trip_branch_flat, d_trip_start, d_trip_count; // ------------------------------------------------------------------------- - // Host-side per-scenario Sbus, ALREADY PERMUTED into active-slot order at - // construction (see class doc). n_scenarios_ is the ORIGINAL (pre- - // compaction) row count — kept for bookkeeping / error messages only; the - // device-resident d_Sbus_all has n_active() rows. + // Device-resident per-scenario Sbus in ACTIVE-slot order (n_active × n_bus), + // filled by set_sbus_from_orig() from the session's original-order buffer. + // n_scenarios_ is the ORIGINAL (pre-compaction) row count — bookkeeping / + // error messages only. // ------------------------------------------------------------------------- - std::vector h_Sbus_all_; int n_scenarios_ = 0; thrust::device_vector d_Sbus_all; // n_active × n_bus thrust::device_vector d_Sbus_batch; // batch_size × n_bus // ------------------------------------------------------------------------- - // set_gen_v() override (see ScenarioSweepSession::set_gen_v's doc), ALREADY - // PERMUTED into active-slot order at construction — same rationale as - // h_Sbus_all_ above. k_active() == 0 (the default) is a cheap no-op. + // set_gen_v() override (see ScenarioSweepSession::set_gen_v's doc), in + // active-slot order. Two ways in: the host path (ctor argument / set_gen_v, + // permuted here) and the device path (set_gen_v_from_orig, one gather + // kernel from the session's original-order device buffer). Only + // gen_v_override_.h_active_bus is meaningful on the device path (its + // k_active() gates prepare_Ybus_batch's reseed); h_gen_v_all stays empty. // ------------------------------------------------------------------------- GenVOverride gen_v_override_; thrust::device_vector d_gv_active_bus; + thrust::device_vector d_gv_active_col; // device path only thrust::device_vector d_gv_all; // Preprocess timing captured at construction (CPU work only). double t_preprocess_ms = 0.0; // ------------------------------------------------------------------------- - // Constructor (host-only): resolve_indices, check_connectivity, and - // build_flat_patches — identical sequence to ContingencyBatch, always the - // legacy (skip-if-split) connectivity path (no MaskConfig / handle_ - // disconnected_grid support for this source, see class doc). Then - // permutes h_Sbus_all_orig into active-slot order. + // Constructor (host-only): resolve_indices, check_connectivity / + // compute_component_masks, and build_flat_patches — identical sequence to + // ContingencyBatch — then permutes the optional gen_v / slack-weight rows + // into active-slot order. Sbus is NOT taken here: call set_sbus_from_orig() + // after initialize() (the session does). // // contingencies : modified in-place (triplets sorted/merged; .disconnected - // set) — one entry per scenario, row-aligned with - // h_Sbus_all_orig. - // h_Sbus_all_orig : (n_scenarios × n_bus) per-unit complex Sbus, ORIGINAL - // (pre-compaction) row order, row-aligned with - // `contingencies`. Taken by rvalue reference since the - // caller has no further use for it; read-only here - // (copied, row-permuted, into h_Sbus_all_), not moved. + // set) — one entry per scenario. // max_batch_size : upper bound on systems per chunk (the user batch_size). - // mask_cfg : handle_disconnected_grid mode when non-null (see - // class doc); nullptr selects the legacy - // check_connectivity skip-if-split path. - // gen_v_override_orig : optional set_gen_v() data, ORIGINAL (pre- - // compaction) row order, row-aligned with - // h_Sbus_all_orig — permuted into active-slot order - // below, same as h_Sbus_all_orig itself. // mask_cfg : the session's mask configuration (ALWAYS given: its // row_info also drives the per-row PV pins, which do // not depend on handle_disconnected_grid). // mask_mode : handle_disconnected_grid mode when true (see class // doc); false selects the legacy check_connectivity // skip-if-split path. + // gen_v_override_orig : optional set_gen_v() data, ORIGINAL (pre- + // compaction) row order — permuted into active-slot + // order below. // h_slack_w_orig : optional per-row slack weights, [n_scenarios * // n_slack] in ORIGINAL row order (empty: every row // keeps the base weights) -- permuted below too. + // forced_batch_size : 0 → rebalance used_batch_size_ over the active + // count (cold path); > 0 → use exactly this chunk + // size (a live driver's capacity, warm path; also the + // session's fixed_batch_capacity mode). // ------------------------------------------------------------------------- ScenarioSweepBatch(std::vector& contingencies, const int* Ybus_rm_outer, const int* Ybus_rm_inner, const Eigen::SparseMatrix& Ybus_rm, - std::vector&& h_Sbus_all_orig, int max_batch_size, const MaskConfig& mask_cfg, bool mask_mode, GenVOverride&& gen_v_override_orig = GenVOverride{}, std::vector&& h_slack_w_orig = std::vector{}, - int n_slack = 0) + int n_slack = 0, + int forced_batch_size = 0) { auto t_start = std::chrono::steady_clock::now(); n_total_ = static_cast(contingencies.size()); n_scenarios_ = n_total_; + n_bus_ = static_cast(Ybus_rm.rows()); resolve_indices(contingencies, Ybus_rm_outer, Ybus_rm_inner); mask_mode_ = mask_mode; @@ -217,8 +234,12 @@ struct ScenarioSweepBatch { int n_active = 0; for (const auto& ctg : contingencies) if (!ctg.disconnected) ++n_active; - const int n_chunks = (n_active + max_batch_size - 1) / max_batch_size; - used_batch_size_ = n_chunks > 0 ? (n_active + n_chunks - 1) / n_chunks : 1; + if (forced_batch_size > 0) { + used_batch_size_ = forced_batch_size; + } else { + const int n_chunks = (n_active + max_batch_size - 1) / max_batch_size; + used_batch_size_ = n_chunks > 0 ? (n_active + n_chunks - 1) / n_chunks : 1; + } build_flat_patches(contingencies, used_batch_size_, h_flat_ctg_id_, h_flat_k_, @@ -237,35 +258,11 @@ struct ScenarioSweepBatch { build_tripped_branch_table(contingencies, active_to_orig_, h_trip_branch_flat_, h_trip_start_, h_trip_count_); - // Permute Sbus rows into active-slot order so prepare_Sbus_batch's - // contiguous-slice copy (verbatim from InjectionBatch) stays correct - // even though the driver's chunk loop runs over active-slot space. - const int n_bus = static_cast(Ybus_rm.rows()); - h_Sbus_all_.resize(static_cast(active_to_orig_.size()) * n_bus); - for (size_t slot = 0; slot < active_to_orig_.size(); ++slot) { - const int orig = active_to_orig_[slot]; - std::copy( - h_Sbus_all_orig.begin() + static_cast(orig) * n_bus, - h_Sbus_all_orig.begin() + static_cast(orig + 1) * n_bus, - h_Sbus_all_.begin() + static_cast(slot) * n_bus); - } - - // Permute set_gen_v() rows into active-slot order too, mirroring - // h_Sbus_all_ above (same active_to_orig_ mapping) — see + // Permute set_gen_v() rows into active-slot order (host path) — see // GenVOverride's own doc. The active-bus column list itself is // row-independent, so it carries over unchanged. - if (gen_v_override_orig.k_active() > 0) { - const ptrdiff_t k = gen_v_override_orig.k_active(); - gen_v_override_.h_active_bus = std::move(gen_v_override_orig.h_active_bus); - gen_v_override_.h_gen_v_all.resize(active_to_orig_.size() * static_cast(k)); - for (size_t slot = 0; slot < active_to_orig_.size(); ++slot) { - const int orig = active_to_orig_[slot]; - std::copy( - gen_v_override_orig.h_gen_v_all.begin() + static_cast(orig) * k, - gen_v_override_orig.h_gen_v_all.begin() + static_cast(orig + 1) * k, - gen_v_override_.h_gen_v_all.begin() + static_cast(slot) * k); - } - } + if (gen_v_override_orig.k_active() > 0) + _permute_gen_v_rows(std::move(gen_v_override_orig)); // Permute the per-row slack weights into active-slot order too. n_slack_ = n_slack; @@ -288,21 +285,49 @@ struct ScenarioSweepBatch { } int used_batch_size() const { return used_batch_size_; } + const std::vector& active_to_orig() const { return active_to_orig_; } + const int* d_active_to_orig_ptr() const { + return d_active_to_orig.empty() ? nullptr + : thrust::raw_pointer_cast(d_active_to_orig.data()); + } ScenarioSweepBatch(ScenarioSweepBatch&&) noexcept = default; + ScenarioSweepBatch& operator=(ScenarioSweepBatch&&) noexcept = default; ScenarioSweepBatch(const ScenarioSweepBatch&) = delete; ScenarioSweepBatch& operator=(const ScenarioSweepBatch&) = delete; - ScenarioSweepBatch& operator=(ScenarioSweepBatch&&) = delete; // ------------------------------------------------------------------------- - // initialize — upload flat Ybus-patch arrays AND the (already active- - // permuted) Sbus rows; allocate the per-chunk Sbus buffer. Unlike - // InjectionBatch, Ybus is NOT tiled once here — it varies per scenario, so - // it is re-tiled + patched every chunk in prepare_Ybus_batch (like - // ContingencyBatch). + // initialize — upload flat Ybus-patch arrays, masks, the tripped table and + // the active map; size the Sbus buffers (values come from + // set_sbus_from_orig()). Unlike InjectionBatch, Ybus is NOT tiled once + // here — it varies per scenario, so it is re-tiled + patched every chunk + // in prepare_Ybus_batch (like ContingencyBatch). // ------------------------------------------------------------------------- void initialize(BatchPfDriverContext& ctx, cudaStream_t cs); + // ------------------------------------------------------------------------- + // set_sbus_from_orig — gather the session's ORIGINAL-row-order per-unit + // Sbus (n_scenarios × n_bus, device) into d_Sbus_all (active-slot order). + // Requires initialize() (d_active_to_orig uploaded, d_Sbus_all sized). + // ------------------------------------------------------------------------- + void set_sbus_from_orig(const cudaComplexType* d_Sbus_orig, cudaStream_t cs); + + // ------------------------------------------------------------------------- + // set_gen_v — host path hot update: permute the original-order override + // into active-slot order and upload (replaces whatever was configured). + // set_gen_v_from_orig — device path: gather the selected generator columns + // of the session's original-order (n_scenarios × n_gen) device matrix + // into active-slot order with one kernel. active_cols/active_bus are the + // Vm-fixed generator columns and their buses (GenVOverride's filter). + // clear_gen_v — drop any override (rows keep the base-case voltage). + // ------------------------------------------------------------------------- + void set_gen_v(GenVOverride&& gen_v_override_orig, cudaStream_t cs); + void set_gen_v_from_orig(const cuda_real_type* d_gen_v_orig, int n_gen, + const std::vector& active_cols, + const std::vector& active_bus, + cudaStream_t cs); + void clear_gen_v() { gen_v_override_ = GenVOverride{}; } + // ------------------------------------------------------------------------- // fill_mask_buffers — write this chunk's handle_disconnected_grid mask // slice into the NrIterBuffers. Sets null / 0 when the mode is off or the @@ -381,6 +406,23 @@ struct ScenarioSweepBatch { } double cpu_preprocess_ms() const { return t_preprocess_ms; } + +private: + // Host permutation of an original-order override into active-slot order + // (h_gen_v_all rows follow active_to_orig_; the bus list is row-independent). + void _permute_gen_v_rows(GenVOverride&& orig) + { + const ptrdiff_t k = orig.k_active(); + gen_v_override_.h_active_bus = std::move(orig.h_active_bus); + gen_v_override_.h_gen_v_all.resize(active_to_orig_.size() * static_cast(k)); + for (size_t slot = 0; slot < active_to_orig_.size(); ++slot) { + const int o = active_to_orig_[slot]; + std::copy( + orig.h_gen_v_all.begin() + static_cast(o) * k, + orig.h_gen_v_all.begin() + static_cast(o + 1) * k, + gen_v_override_.h_gen_v_all.begin() + static_cast(slot) * k); + } + } }; #endif // SCENARIO_SWEEP_BATCH_CUH diff --git a/src/_cpp/contingency/driver.cuh b/src/_cpp/contingency/driver.cuh index 8df8655..bd3de09 100644 --- a/src/_cpp/contingency/driver.cuh +++ b/src/_cpp/contingency/driver.cuh @@ -110,30 +110,17 @@ inline void run_nr_loop( // needs_fresh_jacobian = true → fill every iteration // needs_iter0_jacobian = true → fill only at iter == 0 // both false → never fill (policy reuses base factors) + // The fill sequence itself (zero / dS fill / feature stamps / per-slot + // overrides + mask) lives in nr_fill_J_at_current_V (nr_iter_step.cuh), + // shared with the post-loop converged-J refill of the batched adjoint. if constexpr (Policy::needs_fresh_jacobian) { timer.start(); - nr_feature_zero_J(buf, nnz_J, batch_size, cs); - fill_J_kernel<<>>( - buf.d_J_values, buf.d_V, buf.d_Ibus, - buf.d_Ybus_outer, buf.d_Ybus_inner, buf.d_Ybus_values, - buf.d_map_j11, buf.d_map_j12, buf.d_map_j21, buf.d_map_j22, - n_bus, nnz_Y, nnz_J, batch_size); - nr_feature_fill_J(buf, n_bus, nnz_J, batch_size, cs); - // Per-slot overrides + mask AFTER the feature stamps so they win. - nr_apply_J_masks(buf, nnz_J, dim_J, batch_size, cs); + nr_fill_J_at_current_V(buf, n_bus, dim_J, nnz_Y, nnz_J, batch_size, cs); step.t_fill_J = timer.stop_ms(); } else if constexpr (Policy::needs_iter0_jacobian) { if (iter == 0) { timer.start(); - nr_feature_zero_J(buf, nnz_J, batch_size, cs); - fill_J_kernel<<>>( - buf.d_J_values, buf.d_V, buf.d_Ibus, - buf.d_Ybus_outer, buf.d_Ybus_inner, buf.d_Ybus_values, - buf.d_map_j11, buf.d_map_j12, buf.d_map_j21, buf.d_map_j22, - n_bus, nnz_Y, nnz_J, batch_size); - nr_feature_fill_J(buf, n_bus, nnz_J, batch_size, cs); - // Per-slot overrides + mask AFTER the feature stamps so they win. - nr_apply_J_masks(buf, nnz_J, dim_J, batch_size, cs); + nr_fill_J_at_current_V(buf, n_bus, dim_J, nnz_Y, nnz_J, batch_size, cs); step.t_fill_J = timer.stop_ms(); } } diff --git a/src/_cpp/dlpack_export.cu b/src/_cpp/dlpack_export.cu index 0d6c3a6..37c471d 100644 --- a/src/_cpp/dlpack_export.cu +++ b/src/_cpp/dlpack_export.cu @@ -34,7 +34,10 @@ #include "contingency/batch_sources/scenario_sweep_batch.cuh" #include +#include #include +#include +#include // ============================================================================= // Capsule destructor — registered as the PyCapsule destructor. @@ -209,4 +212,213 @@ export_v_results_dlpack_ss(std::shared_ptr self) auto owner = std::static_pointer_cast(self); auto* mt = make_dl_tensor(ptr, dev, ns, nb, std::move(owner)); return pybind11::capsule(mt, "dltensor", capsule_destructor); +} + +// ============================================================================= +// ScenarioSweepSession differentiable-path importers / exporters +// ============================================================================= + +namespace { + +// A borrowed view of an incoming DLPack capsule: validated pointer + the +// managed tensor, consumed (renamed + deleter run) by release(). +struct DLInput { + DLManagedTensor* mt = nullptr; + const void* data = nullptr; + PyObject* capsule = nullptr; + + void release() { + if (!mt) return; + // Protocol: the consumer renames the capsule so the producer's + // destructor never double-frees, then calls the deleter itself. + PyCapsule_SetName(capsule, "used_dltensor"); + if (mt->deleter) mt->deleter(mt); + mt = nullptr; + } +}; + +// Validate a "dltensor" capsule: on this device, dtype (code, bits), ndim and +// shape (a -1 entry accepts any extent), compact row-major (strides null or +// matching). Returns the view; the caller must release() it once the data +// has been consumed (host-synchronized copy). +DLInput check_dl_input(pybind11::handle capsule, const char* what, + int device_id, uint8_t dtype_code, int dtype_bits, + std::initializer_list shape) +{ + DLInput in; + in.capsule = capsule.ptr(); + if (!PyCapsule_IsValid(in.capsule, "dltensor")) + throw std::runtime_error( + std::string(what) + ": invalid or already-consumed DLPack capsule " + "(expected name \"dltensor\")"); + in.mt = static_cast(PyCapsule_GetPointer(in.capsule, "dltensor")); + const DLTensor& t = in.mt->dl_tensor; + if (t.device.device_type != kDLCUDA || t.device.device_id != device_id) + throw std::runtime_error( + std::string(what) + ": the tensor must live on CUDA device " + + std::to_string(device_id) + " (the session's device)"); + if (t.dtype.code != dtype_code || t.dtype.bits != dtype_bits || t.dtype.lanes != 1) + throw std::runtime_error( + std::string(what) + ": wrong dtype -- expected " + + (dtype_code == kDLComplex ? "complex" : "float") + + std::to_string(dtype_bits) + " (this build's precision)"); + const int ndim = static_cast(shape.size()); + if (t.ndim != ndim) + throw std::runtime_error( + std::string(what) + ": expected a " + std::to_string(ndim) + "-D tensor, got " + + std::to_string(t.ndim) + "-D"); + int d = 0; + for (int64_t expected : shape) { + if (expected >= 0 && t.shape[d] != expected) + throw std::runtime_error( + std::string(what) + ": dimension " + std::to_string(d) + " has extent " + + std::to_string(t.shape[d]) + ", expected " + std::to_string(expected)); + ++d; + } + if (t.strides) { + int64_t expected = 1; + for (int k = t.ndim - 1; k >= 0; --k) { + if (t.shape[k] > 1 && t.strides[k] != expected) + throw std::runtime_error( + std::string(what) + ": the tensor must be contiguous (row-major); " + "call .contiguous() first"); + expected *= t.shape[k]; + } + } + in.data = static_cast(t.data) + t.byte_offset; + return in; +} + +} // namespace + +void import_injections_dlpack_ss(std::shared_ptr self, + pybind11::capsule capsule, + std::uintptr_t producer_stream) +{ + const int dev = self->base_state_->device_id_; + const int n_bus = self->base_state_->n_bus; + DLInput in = check_dl_input(capsule, "set_injections_dlpack", dev, + kDLComplex, kDLPackComplexBits, {-1, n_bus}); + const int n_scen = static_cast(in.mt->dl_tensor.shape[0]); + try { + self->set_injections_device(in.data, n_scen, n_bus, producer_stream); + } catch (...) { + in.release(); + throw; + } + in.release(); +} + +void import_gen_v_dlpack_ss(std::shared_ptr self, + pybind11::capsule capsule, + Eigen::Ref gen_bus, + std::uintptr_t producer_stream) +{ + const int dev = self->base_state_->device_id_; + const int n_gen = static_cast(gen_bus.size()); + DLInput in = check_dl_input(capsule, "set_gen_v_dlpack", dev, + kDLFloat, kDLPackRealBits, {-1, n_gen}); + const int n_scen = static_cast(in.mt->dl_tensor.shape[0]); + try { + self->set_gen_v_device(in.data, n_scen, n_gen, gen_bus, producer_stream); + } catch (...) { + in.release(); + throw; + } + in.release(); +} + +pybind11::tuple +export_solve_jt_batch_dlpack_ss(std::shared_ptr self, + pybind11::capsule rhs_capsule, + pybind11::object j_values_capsule, + pybind11::object ybus_values_capsule, + pybind11::object v_capsule, + bool want_gen_v_grad, + std::uintptr_t producer_stream) +{ + if (!self->solver_) + throw std::runtime_error( + "ScenarioSweepSession: run() must be called before solve_JT_batch_dlpack()"); + const int dev = self->base_state_->device_id_; + const int n_scen = self->solver_->n_contingencies_(); + const int n_bus = self->base_state_->n_bus; + const int dim_J = self->base_state_->dim_J; + const int nnz_J = self->base_state_->nnz_J; + const int nnz_Y = self->base_state_->nnz_Y; + const int cap = self->solver_->batch_size_; + + DLInput rhs = check_dl_input(rhs_capsule, "solve_JT_batch_dlpack(rhs)", dev, + kDLFloat, kDLPackRealBits, {n_scen, dim_J}); + DLInput jv, yv, vv; + std::vector inputs{&rhs}; + try { + if (!j_values_capsule.is_none()) { + jv = check_dl_input(j_values_capsule, "solve_JT_batch_dlpack(j_values)", dev, + kDLFloat, kDLPackRealBits, {cap, nnz_J}); + inputs.push_back(&jv); + } + if (!ybus_values_capsule.is_none()) { + yv = check_dl_input(ybus_values_capsule, "solve_JT_batch_dlpack(ybus_values)", dev, + kDLComplex, kDLPackComplexBits, {cap, nnz_Y}); + inputs.push_back(&yv); + } + if (!v_capsule.is_none()) { + vv = check_dl_input(v_capsule, "solve_JT_batch_dlpack(v)", dev, + kDLComplex, kDLPackComplexBits, {n_scen, n_bus}); + inputs.push_back(&vv); + } + self->solve_JT_batch(rhs.data, jv.data, want_gen_v_grad, yv.data, vv.data, + producer_stream); + } catch (...) { + for (DLInput* p : inputs) p->release(); + throw; + } + // solve_JT_batch host-synchronizes before returning: the inputs may go. + for (DLInput* p : inputs) p->release(); + + auto owner = std::static_pointer_cast(self); + void* lam_ptr = const_cast( + static_cast(self->solver_->d_JT_sol_full_ptr())); + auto* lam_mt = make_dl_tensor_real(lam_ptr, dev, n_scen, owner, dim_J); + pybind11::capsule lam_cap(lam_mt, "dltensor", capsule_destructor); + + pybind11::object gvm_obj = pybind11::none(); + if (want_gen_v_grad) { + void* gvm_ptr = const_cast( + static_cast(self->solver_->d_gvm_full_ptr())); + auto* gvm_mt = make_dl_tensor_real(gvm_ptr, dev, n_scen, owner, n_bus); + gvm_obj = pybind11::capsule(gvm_mt, "dltensor", capsule_destructor); + } + return pybind11::make_tuple(lam_cap, gvm_obj); +} + +pybind11::capsule +export_j_values_dlpack_ss(std::shared_ptr self) +{ + if (!self->solver_) + throw std::runtime_error( + "ScenarioSweepSession: run() must be called before j_values_dlpack()"); + self->solver_->synchronize(); + void* ptr = const_cast(static_cast(self->solver_->j_values_ptr())); + int dev = self->solver_->device_id(); + auto owner = std::static_pointer_cast(self); + auto* mt = make_dl_tensor_real(ptr, dev, self->solver_->batch_size_, std::move(owner), + self->base_state_->nnz_J); + return pybind11::capsule(mt, "dltensor", capsule_destructor); +} + +pybind11::capsule +export_ybus_values_dlpack_ss(std::shared_ptr self) +{ + if (!self->solver_) + throw std::runtime_error( + "ScenarioSweepSession: run() must be called before ybus_values_dlpack()"); + self->solver_->synchronize(); + void* ptr = const_cast(static_cast(self->solver_->ybus_values_ptr())); + int dev = self->solver_->device_id(); + auto owner = std::static_pointer_cast(self); + auto* mt = make_dl_tensor(ptr, dev, self->solver_->batch_size_, + self->base_state_->nnz_Y, std::move(owner)); + return pybind11::capsule(mt, "dltensor", capsule_destructor); } \ No newline at end of file diff --git a/src/_cpp/dlpack_export.cuh b/src/_cpp/dlpack_export.cuh index 1938e02..b3ffa7e 100644 --- a/src/_cpp/dlpack_export.cuh +++ b/src/_cpp/dlpack_export.cuh @@ -92,23 +92,24 @@ inline DLManagedTensor* make_dl_tensor( // make_dl_tensor_real // Real-valued (float32 or float64) variant of make_dl_tensor. // Used for solve_JT_dlpack where the solution vector is real, not complex. -// Always 1-D (no shape_1 parameter). +// shape_1 == 0 → 1-D; otherwise 2-D [shape_0, shape_1] compact row-major. // ============================================================================= inline DLManagedTensor* make_dl_tensor_real( void* data, int device_id, int64_t n, - std::shared_ptr owner) + std::shared_ptr owner, + int64_t shape_1 = 0) { auto* ctx = new DLPackCtx; ctx->owner = std::move(owner); ctx->shape[0] = n; - ctx->shape[1] = 0; + ctx->shape[1] = shape_1; auto* mt = new DLManagedTensor; mt->dl_tensor.data = data; mt->dl_tensor.device = {kDLCUDA, device_id}; - mt->dl_tensor.ndim = 1; + mt->dl_tensor.ndim = (shape_1 == 0) ? 1 : 2; mt->dl_tensor.dtype = {kDLFloat, static_cast(kDLPackRealBits), 1}; diff --git a/src/_cpp/dlpack_export.hpp b/src/_cpp/dlpack_export.hpp index 8eeee55..3425172 100644 --- a/src/_cpp/dlpack_export.hpp +++ b/src/_cpp/dlpack_export.hpp @@ -11,8 +11,11 @@ #pragma once #include +#include #include +#include "Eigen/Core" + // Forward-declare session types — complete type not needed here. struct AcPfNrSession; struct ContingencyAnalysisSession; @@ -53,4 +56,47 @@ pybind11::capsule export_v_base_dlpack_ss( std::shared_ptr self); pybind11::capsule export_v_results_dlpack_ss( + std::shared_ptr self); + +// ----------------------------------------------------------------------------- +// ScenarioSweepSession differentiable-path interop (see _batch_pf.py). +// +// Importers CONSUME the capsule (renamed to "used_dltensor", deleter called +// once the D2D copy is host-synchronized), validate dtype / device / shape / +// contiguity, and hand the device pointer to the session. producer_stream is +// the cudaStream_t handle (as an integer) the tensor was produced on -- +// torch.cuda.current_stream().cuda_stream -- or 0. +// ----------------------------------------------------------------------------- +// (n_scenarios, n_bus) complex, per-unit Sbus. +void import_injections_dlpack_ss( + std::shared_ptr self, + pybind11::capsule capsule, + std::uintptr_t producer_stream); + +// (n_scenarios, n_gen) real vm_pu; gen_bus (n_gen,) AC-solver bus per generator. +void import_gen_v_dlpack_ss( + std::shared_ptr self, + pybind11::capsule capsule, + Eigen::Ref gen_bus, + std::uintptr_t producer_stream); + +// Batched adjoint: rhs (n_scenarios, dim_J) real; optional snapshots +// j_values (capacity, nnz_J) real, ybus_values (capacity, nnz_Y) complex, +// v (n_scenarios, n_bus) complex (pybind11::none() when absent). Returns +// (lambda capsule [n_scenarios, dim_J], gvm capsule [n_scenarios, n_bus] or +// None); both alias driver buffers overwritten by the next call -- clone. +pybind11::tuple export_solve_jt_batch_dlpack_ss( + std::shared_ptr self, + pybind11::capsule rhs_capsule, + pybind11::object j_values_capsule, + pybind11::object ybus_values_capsule, + pybind11::object v_capsule, + bool want_gen_v_grad, + std::uintptr_t producer_stream); + +// Chunk-buffer aliases for the snapshot mode: [capacity, nnz_J] real / +// [capacity, nnz_Y] complex, active-slot order. Clone before the next run(). +pybind11::capsule export_j_values_dlpack_ss( + std::shared_ptr self); +pybind11::capsule export_ybus_values_dlpack_ss( std::shared_ptr self); \ No newline at end of file diff --git a/src/_cpp/nr_iter_step.cuh b/src/_cpp/nr_iter_step.cuh index c57f5ab..9e117a4 100644 --- a/src/_cpp/nr_iter_step.cuh +++ b/src/_cpp/nr_iter_step.cuh @@ -377,6 +377,35 @@ inline void nr_apply_J_masks(const NrIterBuffers& buf, nr_apply_bus_mask(buf, nnz_J, dim_J, batch, cs); } +// ----------------------------------------------------------------------------- +// nr_fill_J_at_current_V +// +// Step ③ alone: the numeric Jacobian at whatever d_V / d_Ibus currently hold +// (d_Ibus must be Ybus · d_V, i.e. the SpMV must have run since the last V +// update). The single definition of the fill sequence -- zero (additive +// features), dS/dx fill, feature stamps, per-slot overrides + bus mask -- used +// by the single-system step, the batched NR loop (driver.cuh), and the +// post-loop refill of J at the CONVERGED V that the batched adjoint needs +// (BatchPfDriver::keep_final_jacobian_). +// ----------------------------------------------------------------------------- +inline void nr_fill_J_at_current_V( + const NrIterBuffers& buf, + int n_bus, int dim_J, int nnz_Y, int nnz_J, + int batch, + cudaStream_t cs) +{ + nr_feature_zero_J(buf, nnz_J, batch, cs); + fill_J_kernel<<>>( + buf.d_J_values, buf.d_V, buf.d_Ibus, + buf.d_Ybus_outer, buf.d_Ybus_inner, buf.d_Ybus_values, + buf.d_map_j11, buf.d_map_j12, buf.d_map_j21, buf.d_map_j22, + n_bus, nnz_Y, nnz_J, batch); + nr_feature_fill_J(buf, n_bus, nnz_J, batch, cs); + // Per-slot overrides + identity-pinned / masked rows win over every stamp + // above (no-op unless the buffers carry mask entries). + nr_apply_J_masks(buf, nnz_J, dim_J, batch, cs); +} + // ----------------------------------------------------------------------------- // nr_iter_step_fill_F // @@ -464,16 +493,7 @@ inline void nr_iter_step_prepare( // When an additive feature (HVDC droop) is active, J must be zeroed first // (the dS fill assigns; the droop slopes accumulate onto / beside it). timer.start(); - nr_feature_zero_J(buf, nnz_J, actual_batch, cs); - fill_J_kernel<<>>( - buf.d_J_values, buf.d_V, buf.d_Ibus, - buf.d_Ybus_outer, buf.d_Ybus_inner, buf.d_Ybus_values, - buf.d_map_j11, buf.d_map_j12, buf.d_map_j21, buf.d_map_j22, - n_bus, nnz_Y, nnz_J, actual_batch); - nr_feature_fill_J(buf, n_bus, nnz_J, actual_batch, cs); - // Identity-pinned / masked rows win over every stamp above (no-op unless - // the buffers carry mask entries). - nr_apply_J_masks(buf, nnz_J, dim_J, actual_batch, cs); + nr_fill_J_at_current_V(buf, n_bus, dim_J, nnz_Y, nnz_J, actual_batch, cs); if (use_cudss) dss_A.set_values(buf.d_J_values); t.t_fill_J += timer.stop_ms(); diff --git a/src/_cpp/python_bindings.cpp b/src/_cpp/python_bindings.cpp index 60e6f0e..118944e 100644 --- a/src/_cpp/python_bindings.cpp +++ b/src/_cpp/python_bindings.cpp @@ -345,6 +345,24 @@ PYBIND11_MODULE(_gpusim2grid, m) "Fixed NR iterations per chunk (no convergence check mid-loop)") .def_readonly("n_refactorize", &BatchTimings::n_refactorize, "Number of REFACTORIZATION calls (n_chunks × nb_iter − 1)") + // --- batched adjoint (ScenarioSweepSession only; cumulative per driver life) --- + .def_readonly("t_adjoint_build_ms", &BatchTimings::t_adjoint_build_ms, + "Wall-clock: transposed-Jacobian skeleton + position map + buffers + " + "cuDSS ANALYSIS of J^T -- paid by the FIRST backward only (ms)") + .def_readonly("t_adjoint_first_factorize", &BatchTimings::t_adjoint_first_factorize, + "cuDSS FACTORIZATION of J^T -- first backward only") + .def_readonly("t_adjoint_refactorize", &BatchTimings::t_adjoint_refactorize, + "cuDSS REFACTORIZATION of J^T -- every later backward after a new run() -- total") + .def_readonly("t_adjoint_solve", &BatchTimings::t_adjoint_solve, + "cuDSS SOLVE with J^T -- every backward -- total") + .def_readonly("adjoint_n_analysis", &BatchTimings::adjoint_n_analysis, + "Number of J^T ANALYSIS calls over the driver's life (0 or 1)") + .def_readonly("adjoint_n_factorize", &BatchTimings::adjoint_n_factorize, + "Number of J^T FACTORIZATION calls over the driver's life (0 or 1)") + .def_readonly("adjoint_n_refactorize", &BatchTimings::adjoint_n_refactorize, + "Number of J^T REFACTORIZATION calls over the driver's life") + .def_readonly("adjoint_n_solve", &BatchTimings::adjoint_n_solve, + "Number of J^T SOLVE calls over the driver's life") .def_readonly("n_disconnected", &BatchTimings::n_disconnected, "Contingencies skipped because they would disconnect the Ybus graph; " "their residuals are set to NaN in the output. With " @@ -442,6 +460,18 @@ PYBIND11_MODULE(_gpusim2grid, m) d2h["copy_residuals_to_host_ms"] = t.t_copy_residuals_to_host_ms; d2h["copy_violations_to_host_ms"] = t.t_copy_violations_to_host_ms; + // Batched adjoint (differentiable path): cumulative over the driver's + // life, separate from run() -- NOT part of 'total'. + py::dict adjoint; + adjoint["build_ms"] = t.t_adjoint_build_ms; + adjoint["first_factorize"] = entry_dict(t.t_adjoint_first_factorize); + adjoint["refactorize"] = entry_dict(t.t_adjoint_refactorize); + adjoint["solve"] = entry_dict(t.t_adjoint_solve); + adjoint["n_analysis"] = t.adjoint_n_analysis; + adjoint["n_factorize"] = t.adjoint_n_factorize; + adjoint["n_refactorize"] = t.adjoint_n_refactorize; + adjoint["n_solve"] = t.adjoint_n_solve; + py::dict result; result["total"] = t.t_grand_total_ms(); result["cpu_preproc"] = cpu_preproc; @@ -449,6 +479,7 @@ PYBIND11_MODULE(_gpusim2grid, m) result["context_init"] = context_init; result["gpu_compute"] = gpu_compute; result["d2h"] = d2h; + result["adjoint"] = adjoint; return result; }, "Nested dict view of the coarse timing buckets: " "{'total': ms, 'cpu_preproc': {'total': ms, ...}, 'h2d': {'total': ms, ...}, " @@ -1552,7 +1583,111 @@ PYBIND11_MODULE(_gpusim2grid, m) "Syncs the base-case stream before returning.") .def("v_results_dlpack", &export_v_results_dlpack_ss, "Export batch voltages as DLPack capsule, shape [n_scenarios, n_bus].\n" - "Requires run() to have been called. Syncs the solver stream."); + "Requires run() to have been called. Syncs the solver stream. The " + "memory is overwritten IN PLACE by the next run() that reuses the " + "batch driver (same n_scenarios and settings) and freed by one that " + "rebuilds it -- clone the tensor for a snapshot either way.") + // ------------------------------------------------------------------- + // Driver persistence + differentiable path (see the Python + // gpusim2grid.differentiable.BatchPowerFlow wrapper). + // ------------------------------------------------------------------- + .def_readwrite("fixed_batch_capacity", &ScenarioSweepSession::fixed_batch_capacity_, + "When True, batch_size is used verbatim as the batch driver's " + "chunk capacity (no rebalancing over the active count): with " + "batch_size >= n_scenarios the whole batch is always solved as " + "ONE chunk whatever rows get islanded. Needed by the adjoint " + "(keep_final_jacobian). Default False. Takes effect on the " + "next run() (rebuilds the driver when changed).") + .def_readwrite("keep_final_jacobian", &ScenarioSweepSession::keep_final_jacobian_, + "When True, run() refills the batched Jacobian at the CONVERGED " + "voltages after the NR loop (one extra fill_J), so that " + "solve_JT_batch_dlpack() can use it. Requires the batch to be " + "solved in one chunk (see fixed_batch_capacity). Default False.") + .def_readonly("run_counter", &ScenarioSweepSession::run_counter_, + "Number of run() calls so far (an autograd backward checks it " + "against the forward it belongs to).") + .def_readonly("driver_build_counter", &ScenarioSweepSession::driver_build_counter_, + "Number of batch-driver (cold) builds so far: allocation + cuDSS " + "ANALYSIS. Stays constant across run() calls that reuse the driver.") + .def_readonly("source_build_counter", &ScenarioSweepSession::source_build_counter_, + "Number of batch-source builds so far (cold + warm runs: topology " + "preprocessing + patch upload). Constant across hot runs.") + .def_property_readonly("adjoint_ready", &ScenarioSweepSession::adjoint_ready, + "True once the batched transposed system exists (first solve_JT_batch_dlpack()).") + .def_property_readonly("capacity", &ScenarioSweepSession::capacity, + "Live driver's chunk capacity (0 before run()).") + .def_property_readonly("n_active", &ScenarioSweepSession::n_active, + "Rows actually solved by the last run() (n_scenarios minus the islanded ones).") + .def_property_readonly("nnz_J", &ScenarioSweepSession::nnz_J, + "Non-zeros of one (augmented) Jacobian.") + .def_property_readonly("nnz_Y", &ScenarioSweepSession::nnz_Y, + "Non-zeros of Ybus.") + .def_property_readonly("p_row_of_bus", &ScenarioSweepSession::p_row_of_bus, + "Bus-keyed J row of each bus' P equation (length n_bus, -1 if none). " + "Re-read after every run(): set_contingency_gens can grow dim_J.") + .def_property_readonly("q_row_of_bus", &ScenarioSweepSession::q_row_of_bus, + "Bus-keyed J row of each bus' Q equation (length n_bus, -1 if none).") + .def_property_readonly("theta_col_of_bus", &ScenarioSweepSession::theta_col_of_bus, + "Bus-keyed J column of each bus' angle unknown (length n_bus, -1 if none).") + .def_property_readonly("vm_col_of_bus", &ScenarioSweepSession::vm_col_of_bus, + "Bus-keyed J column of each bus' |V| unknown (length n_bus, -1 for a " + "Vm-fixed bus).") + .def_property_readonly("is_vm_fixed_bus", &ScenarioSweepSession::is_vm_fixed_bus, + "(n_bus,) 0/1: the bus' |V| is fixed (pv or slack) -- the only buses " + "set_gen_v() acts on and the only ones with a gen_v gradient.") + .def("get_active_to_orig", &ScenarioSweepSession::get_active_to_orig, + "(n_active,) int: original scenario index of each active batch slot " + "(identity before run() or without islanded rows).") + .def("j_skeleton", &ScenarioSweepSession::j_skeleton, + "(outer, inner) int32 CSR structure of one Jacobian (host copies), " + "for tests / external assembly of j_values_dlpack().") + .def("clear_gen_v", &ScenarioSweepSession::clear_gen_v, + "Drop any set_gen_v() override: every row keeps the base-case voltage " + "again. Takes effect on the next run().") + .def("set_injections_dlpack", &import_injections_dlpack_ss, + pybind11::arg("capsule"), pybind11::arg("producer_stream") = 0, + "Device path of set_injections(): a DLPack capsule of a (n_scenarios, " + "n_bus) contiguous complex tensor (this build's precision) of PER-UNIT " + "Sbus rows (AC-solver bus numbering) on this session's device. One " + "device-to-device copy, host-synchronized before returning; the capsule " + "is consumed. producer_stream: the CUDA stream handle the tensor was " + "produced on (torch.cuda.current_stream().cuda_stream), 0 = default. " + "Fixes n_scenarios.") + .def("set_gen_v_dlpack", &import_gen_v_dlpack_ss, + pybind11::arg("capsule"), pybind11::arg("gen_bus"), + pybind11::arg("producer_stream") = 0, + "Device path of set_gen_v(): (n_scenarios, n_gen) contiguous real " + "tensor of vm_pu (this build's precision) on this device; same " + "semantics as set_gen_v(gen_v, gen_bus). Capsule consumed.") + .def("solve_JT_batch_dlpack", &export_solve_jt_batch_dlpack_ss, + pybind11::arg("rhs"), + pybind11::arg("j_values") = pybind11::none(), + pybind11::arg("ybus_values") = pybind11::none(), + pybind11::arg("v") = pybind11::none(), + pybind11::arg("want_gen_v_grad") = false, + pybind11::arg("producer_stream") = 0, + "Batched adjoint solve J_s^T lambda_s = rhs_s for every scenario s, " + "with the Jacobians at the converged voltages of the last run() " + "(keep_final_jacobian=True) -- or with the j_values / ybus_values / v " + "snapshots taken right after that run (j_values_dlpack(), " + "ybus_values_dlpack(), v_results_dlpack(), cloned). rhs: (n_scenarios, " + "dim_J) real, original row order, non-finite entries treated as 0. " + "Returns (lambda, gvm): lambda (n_scenarios, dim_J); gvm (n_scenarios, " + "n_bus) when want_gen_v_grad else None -- the adjoint contraction of " + "each Vm-fixed bus' dS/dVm column (the indirect part of d/d gen_v, " + "sign included), 0 elsewhere. Rows of islanded scenarios are 0. Both " + "capsules alias driver buffers overwritten by the next call: clone. " + "The first call builds the transposed system (J->J^T position map, " + "buffers, one cuDSS ANALYSIS + FACTORIZATION); later calls only " + "permute values, REFACTORIZE (once per new run()) and SOLVE. The " + "capsules given are consumed.") + .def("j_values_dlpack", &export_j_values_dlpack_ss, + "(capacity, nnz_J) real: the batched Jacobian values of the last chunk " + "(active-slot order; rows >= n_active are phantom base-case copies). " + "Aliases the chunk buffer: clone right after run() for a snapshot.") + .def("ybus_values_dlpack", &export_ybus_values_dlpack_ss, + "(capacity, nnz_Y) complex: the per-slot patched Ybus values of the last " + "chunk (active-slot order). Aliases the chunk buffer: clone for a snapshot."); // ----------------------------------------------------------------- // Zero-copy construction from a solved lightsim2grid LSGrid diff --git a/src/_cpp/scenario_sweep_session.cu b/src/_cpp/scenario_sweep_session.cu index 1b071ef..10fc3c3 100644 --- a/src/_cpp/scenario_sweep_session.cu +++ b/src/_cpp/scenario_sweep_session.cu @@ -33,12 +33,63 @@ static constexpr int SESSION_BS = 256; +// ============================================================================= +// ScenarioSweepDeviceData — the session's canonical ORIGINAL-row-order device +// inputs. Both input paths end here: numpy (set_injections / set_gen_v: host +// build then one H→D upload at run()) and device tensors (set_injections_ +// device / set_gen_v_device: one D2D copy). The batch source then gathers +// rows into active-slot order on the device (ScenarioSweepBatch:: +// set_sbus_from_orig / set_gen_v_from_orig), so no host permutation is ever +// done and a torch-resident input never touches the host. +// ============================================================================= +struct ScenarioSweepDeviceData { + thrust::device_vector d_Sbus_orig; // n_scenarios × n_bus, per-unit + thrust::device_vector d_gen_v_orig; // n_scenarios × n_gen (device path only) + int gen_v_n_gen = 0; + std::vector gv_active_cols; // Vm-fixed generator columns (device path) + std::vector gv_active_bus; // ... and their AC-solver buses +}; + // ============================================================================= // Destructor — defined here where AcPfNrState / BatchPfDriver // are complete types, enabling unique_ptr to call delete correctly. // ============================================================================= ScenarioSweepSession::~ScenarioSweepSession() = default; +void ScenarioSweepSession::_wait_producer(std::uintptr_t producer_stream, void* cs) +{ + if (producer_stream == 0) return; + cudaEvent_t ev = nullptr; + if (cudaEventCreateWithFlags(&ev, cudaEventDisableTiming) != cudaSuccess) + throw std::runtime_error("ScenarioSweepSession: cudaEventCreate failed"); + cudaError_t e1 = cudaEventRecord(ev, reinterpret_cast(producer_stream)); + cudaError_t e2 = cudaStreamWaitEvent(static_cast(cs), ev, 0); + cudaEventDestroy(ev); + if (e1 != cudaSuccess || e2 != cudaSuccess) + throw std::runtime_error( + "ScenarioSweepSession: could not synchronize with the producer stream (" + + std::string(cudaGetErrorString(e1 != cudaSuccess ? e1 : e2)) + ")"); +} + +ScenarioSweepDriverConfig ScenarioSweepSession::_current_config() const +{ + ScenarioSweepDriverConfig c; + c.n_scenarios = n_scenarios_; + c.batch_size = batch_size_; + c.refactor_period = refactor_period_; + c.strategy = strategy_type_; + c.reordering_alg = reordering_alg_; + c.matching_alg = matching_alg_; + c.pivot_epsilon_alg = pivot_epsilon_alg_; + c.scaling_max_voltage_change = scaling_max_voltage_change_; + c.max_dVa = max_dVa_; + c.max_dVm = max_dVm_; + c.mask_mode = handle_disconnected_grid_; + c.fixed_batch_capacity = fixed_batch_capacity_; + c.base_state_generation = base_state_generation_; + return c; +} + // ============================================================================= // Constructor — base-case NR to convergence. // ============================================================================= @@ -87,6 +138,7 @@ ScenarioSweepSession::ScenarioSweepSession( { (void)slack_weights; if (ledger != nullptr) base_ledger_ = std::make_unique(*ledger); + dev_ = std::make_unique(); _build_base_state(std::vector{}); @@ -115,9 +167,11 @@ ScenarioSweepSession::ScenarioSweepSession( void ScenarioSweepSession::_build_base_state(const std::vector& switchable_buses) { // solver_ references *base_state_: drop it first. Its d_V_results / - // DLPack views die with it (run() rebuilds it anyway). + // DLPack views (and any lazily built adjoint) die with it; the next run() + // is a cold one (the generation counter is part of the driver config). solver_.reset(); has_violations_result_ = false; + ++base_state_generation_; const LedgerData* ledger_ptr = nullptr; LedgerData ext; @@ -203,8 +257,9 @@ void ScenarioSweepSession::set_contingency_gens(Eigen::Ref mask) "now: remote voltage control is not supported by this feature yet."); } - gen_off_ = mask; - has_gen_off_ = true; + gen_off_ = mask; + has_gen_off_ = true; + gen_off_dirty_ = true; } int ScenarioSweepSession::dim_J() const @@ -352,9 +407,63 @@ void ScenarioSweepSession::set_gen_v( throw std::runtime_error( "ScenarioSweepSession::set_gen_v: n_scenarios must be > 0"); - gen_v_ = gen_v; - gen_bus_ = gen_bus; - has_gen_v_ = true; + gen_v_ = gen_v; + gen_bus_ = gen_bus; + has_gen_v_ = true; + gen_v_dirty_ = true; + gen_v_on_device_ = false; +} + +void ScenarioSweepSession::set_gen_v_device(const void* d_ptr, int n_scen, int n_gen, + Eigen::Ref gen_bus, + std::uintptr_t producer_stream) +{ + if (n_gen != gen_bus.size()) + throw std::runtime_error( + "ScenarioSweepSession::set_gen_v_device: gen_v's column count must " + "equal gen_bus's length (one entry per generator)"); + if (n_scen <= 0) + throw std::runtime_error( + "ScenarioSweepSession::set_gen_v_device: n_scenarios must be > 0"); + + // Vm-fixed column filter (same rule as build_gen_v_override). + const int n_bus = base_state_->n_bus; + dev_->gv_active_cols.clear(); + dev_->gv_active_bus.clear(); + for (int g = 0; g < n_gen; ++g) { + const int b = gen_bus(g); + if (b >= 0 && b < n_bus && h_is_vm_fixed_bus_[static_cast(b)]) { + dev_->gv_active_cols.push_back(g); + dev_->gv_active_bus.push_back(b); + } + } + dev_->gen_v_n_gen = n_gen; + + cudaStream_t cs = static_cast(base_state_->cs); + _wait_producer(producer_stream, cs); + dev_->d_gen_v_orig.resize(static_cast(n_scen) * n_gen); + cudaError_t e = cudaMemcpyAsync( + thrust::raw_pointer_cast(dev_->d_gen_v_orig.data()), d_ptr, + static_cast(n_scen) * n_gen * sizeof(cuda_real_type), + cudaMemcpyDeviceToDevice, cs); + if (e != cudaSuccess) + throw std::runtime_error(std::string("ScenarioSweepSession::set_gen_v_device: ") + + cudaGetErrorString(e)); + base_state_->cs.synchronize(); + + gen_v_.resize(n_scen, n_gen); // row-count bookkeeping only (run() checks rows) + gen_bus_ = gen_bus; + has_gen_v_ = true; + gen_v_dirty_ = true; + gen_v_on_device_ = true; +} + +void ScenarioSweepSession::clear_gen_v() +{ + if (!has_gen_v_) return; + has_gen_v_ = false; + gen_v_dirty_ = true; + gen_v_on_device_ = false; } // ============================================================================= @@ -409,6 +518,38 @@ void ScenarioSweepSession::set_injections( sn_mva_ = sn_mva; n_scenarios_ = static_cast(p_mw.rows()); has_injections_ = true; + injections_dirty_ = true; + injections_on_device_ = false; +} + +void ScenarioSweepSession::set_injections_device(const void* d_ptr, int n_scen, int n_bus, + std::uintptr_t producer_stream) +{ + if (n_bus != base_state_->n_bus) + throw std::runtime_error( + "ScenarioSweepSession::set_injections_device: second dim must equal n_bus"); + if (n_scen <= 0) + throw std::runtime_error( + "ScenarioSweepSession::set_injections_device: n_scenarios must be > 0"); + + cudaStream_t cs = static_cast(base_state_->cs); + _wait_producer(producer_stream, cs); + dev_->d_Sbus_orig.resize(static_cast(n_scen) * n_bus); + cudaError_t e = cudaMemcpyAsync( + thrust::raw_pointer_cast(dev_->d_Sbus_orig.data()), d_ptr, + static_cast(n_scen) * n_bus * sizeof(cudaComplexType), + cudaMemcpyDeviceToDevice, cs); + if (e != cudaSuccess) + throw std::runtime_error(std::string("ScenarioSweepSession::set_injections_device: ") + + cudaGetErrorString(e)); + // Host-sync before returning: the caller's buffer may be a temporary that + // its allocator recycles as soon as we return. + base_state_->cs.synchronize(); + + n_scenarios_ = n_scen; + has_injections_ = true; + injections_dirty_ = true; + injections_on_device_ = true; } // ============================================================================= @@ -438,7 +579,8 @@ void ScenarioSweepSession::set_topology( h_yff_eff_, h_yft_eff_, h_ytf_eff_, h_ytt_eff_)); tripped_branches_per_scenario_.push_back(branch_ids); } - has_topology_ = true; + has_topology_ = true; + topology_dirty_ = true; } // ============================================================================= @@ -564,64 +706,135 @@ void ScenarioSweepSession::run() const int n_bus = base_state_->n_bus; - // Build host-side per-unit complex Sbus_all, ORIGINAL (pre-compaction) row - // order — ScenarioSweepBatch permutes into active-slot order internally. - auto t_sbus_start = std::chrono::steady_clock::now(); - std::vector h_Sbus_all( - static_cast(n_scenarios_) * static_cast(n_bus)); - const double inv_sn = 1.0 / sn_mva_; - for (int s = 0; s < n_scenarios_; ++s) { - for (int b = 0; b < n_bus; ++b) { - h_Sbus_all[static_cast(s) * n_bus + b] = - CudaFunHelper::my_make_cuComplex( - static_cast(p_mw_(s, b) * inv_sn), - static_cast(q_mvar_(s, b) * inv_sn)); + // ------------------------------------------------------------------------- + // Canonical per-unit Sbus on the device (ORIGINAL row order). The numpy + // path builds it on the host from the MW/MVAr matrices and uploads it + // once per changed injection; the device path (set_injections_device) + // already filled it. The batch source gathers it into active-slot order + // below, whatever the path. + // ------------------------------------------------------------------------- + t_sbus_build_ms_ = 0.; + if (injections_dirty_ && !injections_on_device_) { + auto t_sbus_start = std::chrono::steady_clock::now(); + std::vector h_Sbus_all( + static_cast(n_scenarios_) * static_cast(n_bus)); + const double inv_sn = 1.0 / sn_mva_; + for (int s = 0; s < n_scenarios_; ++s) { + for (int b = 0; b < n_bus; ++b) { + h_Sbus_all[static_cast(s) * n_bus + b] = + CudaFunHelper::my_make_cuComplex( + static_cast(p_mw_(s, b) * inv_sn), + static_cast(q_mvar_(s, b) * inv_sn)); + } + } + upload_h2d(dev_->d_Sbus_orig, h_Sbus_all.data(), h_Sbus_all.size(), + static_cast(base_state_->cs)); + base_state_->cs.synchronize(); + t_sbus_build_ms_ = ms_since(t_sbus_start); + } + if (dev_->d_Sbus_orig.size() != static_cast(n_scenarios_) * n_bus) + throw std::runtime_error( + "ScenarioSweepSession: injection buffer size does not match " + "n_scenarios x n_bus (call set_injections() again)"); + + // ------------------------------------------------------------------------- + // Cold / warm / hot path selection against the live driver. + // ------------------------------------------------------------------------- + const ScenarioSweepDriverConfig cfg = _current_config(); + const bool cold = !solver_ || cfg != driver_cfg_; + const bool warm = !cold && (topology_dirty_ || gen_off_dirty_); + + if (cold) { + // Host preprocessing (resolve_indices + connectivity/masking + + // build_flat_patches), mutates contingencies_ in-place so the + // disconnected flags are observable below. gen_v is set afterwards + // through the same setter every path uses. + ScenarioSweepBatch source( + contingencies_, + Ybus_rm_.outerIndexPtr(), + Ybus_rm_.innerIndexPtr(), + Ybus_rm_, + batch_size_, + mask_cfg_, + handle_disconnected_grid_, + GenVOverride{}, + std::move(h_slack_w_rows), + base_state_->n_slack, + /*forced_batch_size=*/fixed_batch_capacity_ ? batch_size_ : 0); + used_batch_size_ = source.used_batch_size(); + + // New driver: allocation, block-diagonal structure, cuDSS ANALYSIS. + solver_ = std::make_unique( + *base_state_, + std::move(source), + n_scenarios_, + used_batch_size_, + nb_iter_, + strategy_type_, + refactor_period_, + reordering_alg_, + matching_alg_, + pivot_epsilon_alg_, + scaling_max_voltage_change_, + max_dVa_, + max_dVm_); + driver_cfg_ = cfg; + ++driver_build_counter_; + ++source_build_counter_; + } else if (warm) { + // New topology / generator mask on the live driver: rebuild only the + // source (CPU connectivity + patches + H→D of those) at the driver's + // capacity; no analysis, no allocation of the chunk buffers. + ScenarioSweepBatch source( + contingencies_, + Ybus_rm_.outerIndexPtr(), + Ybus_rm_.innerIndexPtr(), + Ybus_rm_, + batch_size_, + mask_cfg_, + handle_disconnected_grid_, + GenVOverride{}, + std::move(h_slack_w_rows), + base_state_->n_slack, + /*forced_batch_size=*/solver_->batch_size_); + solver_->replace_source(std::move(source)); + used_batch_size_ = solver_->batch_size_; + solver_->mark_reused(/*hot=*/false); + ++source_build_counter_; + } else { + // Hot: only injections / gen_v changed (or nothing at all). + solver_->mark_reused(/*hot=*/true); + } + + // Sbus rows: original order → active slots, one gather on every path. + const cudaStream_t scs = static_cast(solver_->cs); + solver_->source_.set_sbus_from_orig( + thrust::raw_pointer_cast(dev_->d_Sbus_orig.data()), scs); + + // gen_v override: (re)applied whenever the source is new or gen_v changed. + if (cold || warm || gen_v_dirty_) { + if (!has_gen_v_) { + solver_->source_.clear_gen_v(); + } else if (gen_v_on_device_) { + solver_->source_.set_gen_v_from_orig( + thrust::raw_pointer_cast(dev_->d_gen_v_orig.data()), + dev_->gen_v_n_gen, dev_->gv_active_cols, dev_->gv_active_bus, scs); + } else { + solver_->source_.set_gen_v( + build_gen_v_override(gen_v_, gen_bus_, h_is_vm_fixed_bus_), scs); } } - t_sbus_build_ms_ = ms_since(t_sbus_start); - - // Host preprocessing (resolve_indices + connectivity/masking + - // build_flat_patches + Sbus active-order permute), mutates contingencies_ - // in-place so disconnected flags are observable below. - GenVOverride gen_v_override; - if (has_gen_v_) - gen_v_override = build_gen_v_override(gen_v_, gen_bus_, h_is_vm_fixed_bus_); - - ScenarioSweepBatch source( - contingencies_, - Ybus_rm_.outerIndexPtr(), - Ybus_rm_.innerIndexPtr(), - Ybus_rm_, - std::move(h_Sbus_all), - batch_size_, - mask_cfg_, - handle_disconnected_grid_, - std::move(gen_v_override), - std::move(h_slack_w_rows), - base_state_->n_slack); - used_batch_size_ = source.used_batch_size(); - - // (Re-)construct the solver — allows run() to be called multiple times. - solver_ = std::make_unique( - *base_state_, - std::move(source), - n_scenarios_, - used_batch_size_, - nb_iter_, - strategy_type_, - refactor_period_, - reordering_alg_, - matching_alg_, - pivot_epsilon_alg_, - scaling_max_voltage_change_, - max_dVa_, - max_dVm_); + + // Per-run knobs that need no rebuild. + solver_->nb_iter_ = nb_iter_; + solver_->keep_final_jacobian_ = keep_final_jacobian_; // compute_limit_violations: the fused per-chunk kernel needs branch // admittances + limits on device BEFORE solve() runs its chunk loop // (unlike compute_flows(), which uploads AFTER solve() and only for // callers who explicitly want full flows). Mirrors - // ContingencyAnalysisSession::run() exactly. + // ContingencyAnalysisSession::run() exactly; the admittances are uploaded + // once per driver, the limits (which reset the per-row sentinels) every run. double t_admittance_upload_ms = 0.; double t_limits_setup_ms = 0.; if (compute_limit_violations_) { @@ -634,10 +847,12 @@ void ScenarioSweepSession::run() "ScenarioSweepSession: compute_limit_violations requires " "set_limits() to have been called first."); - solver_->upload_branch_admittances( - h_branch_from_, h_branch_to_, h_yff_eff_, h_yft_eff_, h_ytf_eff_, h_ytt_eff_, - h_bus_vn_kv_, sn_mva_); - t_admittance_upload_ms = solver_->branch_data_upload_ms(); + if (!solver_->_has_branch_admittances) { + solver_->upload_branch_admittances( + h_branch_from_, h_branch_to_, h_yff_eff_, h_yft_eff_, h_ytf_eff_, h_ytt_eff_, + h_bus_vn_kv_, sn_mva_); + t_admittance_upload_ms = solver_->branch_data_upload_ms(); + } solver_->set_violation_limits( h_bus_vmin_kv_, h_bus_vmax_kv_, @@ -647,19 +862,24 @@ void ScenarioSweepSession::run() } timings_ = solver_->solve(); - timings_.t_base_case_ms = t_base_case_ms_; - timings_.t_preprocess_ms += base_state_->timings.t_build_J_ms; - timings_.t_alloc_ms += base_state_->timings.t_upload_ms; + timings_.t_base_case_ms = t_base_case_ms_; + timings_.t_preprocess_ms += t_sbus_build_ms_; + if (cold) { + // Base-state costs belong to the construction that this driver + // amortises; report them with the cold run only. + timings_.t_preprocess_ms += base_state_->timings.t_build_J_ms; + timings_.t_alloc_ms += base_state_->timings.t_upload_ms; + timings_.t_context_init_ms += base_state_->timings.t_context_init_ms; + timings_.t_base_case_solve_only_ms = + t_base_case_ms_ - base_state_->timings.t_build_J_ms + - base_state_->timings.t_upload_ms + - base_state_->timings.t_context_init_ms; + timings_.t_ground_truth_check_ms = base_state_->timings.t_ground_truth_check_ms; + } if (compute_limit_violations_) { timings_.t_branch_data_upload_ms += t_admittance_upload_ms; timings_.t_violation_setup_ms += t_limits_setup_ms; } - timings_.t_context_init_ms += base_state_->timings.t_context_init_ms; - timings_.t_base_case_solve_only_ms = - t_base_case_ms_ - base_state_->timings.t_build_J_ms - - base_state_->timings.t_upload_ms - - base_state_->timings.t_context_init_ms; - timings_.t_ground_truth_check_ms = base_state_->timings.t_ground_truth_check_ms; // Scenarios whose topology change disconnects the grid are compacted out // of the batch by the source and never solved; the driver pre-fills their @@ -670,9 +890,95 @@ void ScenarioSweepSession::run() timings_.n_disconnected = n_disconnected; has_violations_result_ = compute_limit_violations_; + injections_dirty_ = topology_dirty_ = gen_v_dirty_ = gen_off_dirty_ = false; + last_run_kept_jacobian_ = keep_final_jacobian_; + ++run_counter_; + solver_->cs.synchronize(); } +// ============================================================================= +// Batched adjoint + structure accessors +// ============================================================================= +void ScenarioSweepSession::solve_JT_batch(const void* d_rhs_orig, const void* d_J_ext, + bool want_gen_v_grad, + const void* d_Ybus_ext, const void* d_V_ext_orig, + std::uintptr_t producer_stream) +{ + if (!solver_) + throw std::runtime_error( + "ScenarioSweepSession::solve_JT_batch: call run() first"); + if (d_J_ext == nullptr && !last_run_kept_jacobian_) + throw std::runtime_error( + "ScenarioSweepSession::solve_JT_batch: the last run() did not keep " + "the converged Jacobian (set keep_final_jacobian = True before run(), " + "or pass a J snapshot)"); + if (want_gen_v_grad && d_V_ext_orig == nullptr && d_J_ext != nullptr) + throw std::runtime_error( + "ScenarioSweepSession::solve_JT_batch: a J snapshot needs the matching " + "V (and Ybus) snapshots for the gen_v gradient"); + _wait_producer(producer_stream, static_cast(solver_->cs)); + solver_->solve_JT_batch( + static_cast(d_rhs_orig), + static_cast(d_J_ext), + want_gen_v_grad, + static_cast(d_Ybus_ext), + static_cast(d_V_ext_orig), + h_is_vm_fixed_bus_); +} + +bool ScenarioSweepSession::adjoint_ready() const { return solver_ && solver_->adjoint_ready(); } +int ScenarioSweepSession::capacity() const { return solver_ ? solver_->batch_size_ : 0; } +int ScenarioSweepSession::n_active() const { return solver_ ? solver_->n_active_ : 0; } +int ScenarioSweepSession::nnz_J() const { return base_state_ ? base_state_->nnz_J : 0; } +int ScenarioSweepSession::nnz_Y() const { return base_state_ ? base_state_->nnz_Y : 0; } +std::vector ScenarioSweepSession::p_row_of_bus() const { return base_state_->h_p_row_of_bus; } +std::vector ScenarioSweepSession::q_row_of_bus() const { return base_state_->h_q_row_of_bus; } +std::vector ScenarioSweepSession::theta_col_of_bus() const { return base_state_->h_theta_col_of_bus; } +std::vector ScenarioSweepSession::vm_col_of_bus() const { return base_state_->h_vm_col_of_bus; } +std::vector ScenarioSweepSession::is_vm_fixed_bus() const +{ + return std::vector(h_is_vm_fixed_bus_.begin(), h_is_vm_fixed_bus_.end()); +} + +Eigen::VectorXi ScenarioSweepSession::get_active_to_orig() const +{ + if (!solver_) { + Eigen::VectorXi out(n_scenarios_); + for (int i = 0; i < n_scenarios_; ++i) out(i) = i; + return out; + } + const std::vector& a2o = solver_->source_.active_to_orig(); + Eigen::VectorXi out(static_cast(a2o.size())); + for (size_t i = 0; i < a2o.size(); ++i) out(static_cast(i)) = a2o[i]; + return out; +} + +std::pair, std::vector> ScenarioSweepSession::j_skeleton() const +{ + thrust::host_vector h_outer(base_state_->d_J_outer); + thrust::host_vector h_inner(base_state_->d_J_inner); + return {std::vector(h_outer.begin(), h_outer.end()), + std::vector(h_inner.begin(), h_inner.end())}; +} + +BatchTimings ScenarioSweepSession::get_timings() const +{ + BatchTimings t = timings_; + if (solver_ && solver_->adjoint_) { + const BatchAdjoint& A = *solver_->adjoint_; + t.t_adjoint_build_ms = A.t_build_ms; + t.t_adjoint_first_factorize = A.t_first_factorize; + t.t_adjoint_refactorize = A.t_refactorize; + t.t_adjoint_solve = A.t_solve; + t.adjoint_n_analysis = A.n_analysis; + t.adjoint_n_factorize = A.n_factorize; + t.adjoint_n_refactorize = A.n_refactorize; + t.adjoint_n_solve = A.n_solve; + } + return t; +} + // ============================================================================= // compute_flows // ============================================================================= @@ -685,11 +991,15 @@ void ScenarioSweepSession::compute_flows() throw std::runtime_error( "ScenarioSweepSession: call run() before compute_flows()"); - solver_->set_branch_data( - h_branch_from_, h_branch_to_, - h_yff_eff_, h_yft_eff_, h_ytf_eff_, h_ytt_eff_, - h_bus_vn_kv_, sn_mva_); - timings_.t_branch_data_upload_ms += solver_->branch_data_upload_ms(); + // The admittances + dense flow buffers survive on a persistent driver: + // upload them once per driver, not once per compute_flows() call. + if (!solver_->_has_branch_data) { + solver_->set_branch_data( + h_branch_from_, h_branch_to_, + h_yff_eff_, h_yft_eff_, h_ytf_eff_, h_ytt_eff_, + h_bus_vn_kv_, sn_mva_); + timings_.t_branch_data_upload_ms += solver_->branch_data_upload_ms(); + } const int n_scen = solver_->n_contingencies; const int n_bra = solver_->n_branches_; diff --git a/src/_cpp/scenario_sweep_session.hpp b/src/_cpp/scenario_sweep_session.hpp index 8d60f64..f099a5a 100644 --- a/src/_cpp/scenario_sweep_session.hpp +++ b/src/_cpp/scenario_sweep_session.hpp @@ -51,16 +51,55 @@ #include "Eigen/Core" #include "Eigen/SparseCore" +#include #include +#include #include // Forward-declare CUDA-dependent types to keep CUDA headers out of this file. struct AcPfNrState; struct LedgerData; struct ScenarioSweepBatch; +struct ScenarioSweepDeviceData; // session-owned device buffers (defined in the .cu) template struct BatchPfDriver; using ScenarioSweepSolver = BatchPfDriver; +// ============================================================================= +// ScenarioSweepDriverConfig — the construction-time shape of the live batch +// driver. run() compares the current settings against the snapshot taken when +// solver_ was built: any difference means the driver cannot be reused (a +// "cold" run rebuilds it, with a new cuDSS ANALYSIS); otherwise run() only +// swaps the source (new topology, "warm") or the injections ("hot"). The +// config members are plain read/write attributes on the Python side, so a +// snapshot comparison is the only robust way to notice a change. +// ============================================================================= +struct ScenarioSweepDriverConfig { + int n_scenarios = -1; + int batch_size = 0; + int refactor_period = 1; + ContingencySolverType strategy = ContingencySolverType::DirectRefactorEvery; + ReorderingAlg reordering_alg = ReorderingAlg::Default; + MatchingAlg matching_alg = MatchingAlg::None; + PivotEpsilonAlg pivot_epsilon_alg = PivotEpsilonAlg::Default; + bool scaling_max_voltage_change = false; + double max_dVa = 0.5, max_dVm = 0.1; + bool mask_mode = false; // handle_disconnected_grid + bool fixed_batch_capacity = false; + int base_state_generation = -1; + + bool operator==(const ScenarioSweepDriverConfig& o) const { + return n_scenarios == o.n_scenarios && batch_size == o.batch_size + && refactor_period == o.refactor_period && strategy == o.strategy + && reordering_alg == o.reordering_alg && matching_alg == o.matching_alg + && pivot_epsilon_alg == o.pivot_epsilon_alg + && scaling_max_voltage_change == o.scaling_max_voltage_change + && max_dVa == o.max_dVa && max_dVm == o.max_dVm + && mask_mode == o.mask_mode && fixed_batch_capacity == o.fixed_batch_capacity + && base_state_generation == o.base_state_generation; + } + bool operator!=(const ScenarioSweepDriverConfig& o) const { return !(*this == o); } +}; + struct ScenarioSweepSession { // Destructor defined in the .cu where AcPfNrState/ScenarioSweepSolver are @@ -71,7 +110,45 @@ struct ScenarioSweepSession { // Owned GPU state // ========================================================================= std::unique_ptr base_state_; - std::unique_ptr solver_; // null until run() + std::unique_ptr solver_; // null until the first run(); then PERSISTENT + std::unique_ptr dev_; // canonical original-order Sbus / gen_v device buffers + + // ========================================================================= + // Driver persistence (see ScenarioSweepDriverConfig and run()). + // + // *_dirty_ : which inputs changed since the last run(). + // injections_on_device_ / gen_v_on_device_ : the canonical buffer was + // last filled straight from a device tensor + // (set_injections_dlpack / set_gen_v_dlpack), so + // run() must not overwrite it from the host copies. + // fixed_batch_capacity_ : when true, batch_size_ is used verbatim as the + // driver's chunk capacity (no rebalancing over the + // active count), so with batch_size_ >= n_scenarios + // the whole batch is always ONE chunk whatever + // rows get islanded -- what the differentiable + // wrapper needs (the adjoint reads the last chunk's + // Jacobian). Default false keeps the historic + // rebalancing for the plain sweep API. + // keep_final_jacobian_ : refill J at the converged V after the NR loop + // (forwarded to the driver; see BatchPfDriver). + // run_counter_ etc. : observability for callers/tests (a torch + // autograd backward checks run_counter_ against + // the forward it belongs to). + // ========================================================================= + bool injections_dirty_ = false; + bool topology_dirty_ = false; + bool gen_v_dirty_ = false; + bool gen_off_dirty_ = false; + bool injections_on_device_ = false; + bool gen_v_on_device_ = false; + bool fixed_batch_capacity_ = false; + bool keep_final_jacobian_ = false; + bool last_run_kept_jacobian_ = false; + int run_counter_ = 0; + int driver_build_counter_ = 0; + int source_build_counter_ = 0; + int base_state_generation_ = 0; + ScenarioSweepDriverConfig driver_cfg_; // RowMajor Ybus copy — needed to build the block-diag CSR + resolve_indices. Eigen::SparseMatrix Ybus_rm_; @@ -262,6 +339,19 @@ struct ScenarioSweepSession { double sn_mva ); + // ========================================================================= + // set_injections_device — the device path of set_injections(): d_ptr is a + // (n_scen × n_bus) row-major PER-UNIT complex buffer (the build's + // cudaComplexType) on this session's device, e.g. a torch tensor handed + // over through DLPack (see dlpack_export.cu). One D2D copy into the + // canonical original-order buffer; producer_stream (a cudaStream_t + // handle, 0 = none) is waited on through an event first, and the copy is + // host-synchronized before returning so the caller may free/reuse the + // source immediately. Fixes n_scenarios(). + // ========================================================================= + void set_injections_device(const void* d_ptr, int n_scen, int n_bus, + std::uintptr_t producer_stream); + // ========================================================================= // set_topology — one branch-id list per scenario (lines-then-trafos), // row-aligned with set_injections(). Requires set_branch_data() first. @@ -282,6 +372,19 @@ struct ScenarioSweepSession { Eigen::Ref> gen_v, Eigen::Ref gen_bus); + // Device path of set_gen_v(): d_ptr is a (n_scen × n_gen) row-major real + // buffer (the build's cuda_real_type) on this device; the Vm-fixed column + // filter is derived from gen_bus on the host (cheap, O(n_gen)) and the + // selected columns are gathered on the device at run(). Same stream / + // sync contract as set_injections_device. + void set_gen_v_device(const void* d_ptr, int n_scen, int n_gen, + Eigen::Ref gen_bus, + std::uintptr_t producer_stream); + + // Drop any gen_v override: every row keeps the grid's base-case voltage + // again (the state before set_gen_v was ever called). + void clear_gen_v(); + // ========================================================================= // set_gen_contingency_data — per-generator snapshot (bridge factory only; // see gen_contingency_data.hpp). Enables set_contingency_gens(). @@ -315,15 +418,54 @@ struct ScenarioSweepSession { bool has_gen_contingency() const { return has_gen_off_; } // ========================================================================= - // run — constructs BatchPfDriver + runs the chunk - // loop. A scenario whose topology change disconnects the grid is skipped - // (NaN residual/voltage) — unless handle_disconnected_grid_ is set, in - // which case only scenarios stranding the angle reference or a - // controller bus are left as NaN (the rest solve on their largest - // connected component, masked buses reported as NaN). + // run — solves every scenario. A scenario whose topology change + // disconnects the grid is skipped (NaN residual/voltage) — unless + // handle_disconnected_grid_ is set, in which case only scenarios stranding + // the angle reference or a controller bus are left as NaN (the rest solve + // on their largest connected component, masked buses reported as NaN). + // + // Three paths, decided against the live driver (see + // ScenarioSweepDriverConfig): + // cold : no driver yet, or its shape/config changed (n_scenarios, + // batch_size, strategy, cuDSS config, base state, ...) → build a + // new BatchPfDriver (allocation + cuDSS + // ANALYSIS + first FACTORIZATION on the first iteration). + // warm : only the topology / generator mask changed → new + // ScenarioSweepBatch (CPU connectivity + patches) swapped into + // the live driver; no analysis, REFACTORIZATION only. + // hot : only injections / gen_v changed → one device gather of the new + // rows; nothing else touched. + // The result buffers (v_results_dlpack) are then overwritten IN PLACE + // across runs (they only move on a cold rebuild). // ========================================================================= void run(); + // ========================================================================= + // solve_JT_batch — batched adjoint (see BatchPfDriver::solve_JT_batch; + // pointer arguments are device buffers of the documented shapes, nullptr + // where optional). Requires the last run() to have been made with + // keep_final_jacobian_ = true (alias mode) or an external J snapshot. + // Builds the transposed system lazily on the first call. + // ========================================================================= + void solve_JT_batch(const void* d_rhs_orig, const void* d_J_ext, + bool want_gen_v_grad, + const void* d_Ybus_ext, const void* d_V_ext_orig, + std::uintptr_t producer_stream); + + // Adjoint / structure accessors (see the bindings for the shapes). + bool adjoint_ready() const; + int capacity() const; // live driver's chunk capacity (0 before run()) + int n_active() const; // rows actually solved by the last run() + int nnz_J() const; + int nnz_Y() const; + std::vector p_row_of_bus() const; + std::vector q_row_of_bus() const; + std::vector theta_col_of_bus() const; + std::vector vm_col_of_bus() const; + std::vector is_vm_fixed_bus() const; + Eigen::VectorXi get_active_to_orig() const; + std::pair, std::vector> j_skeleton() const; // (outer, inner) + // ========================================================================= // compute_flows — branch flows for ALL scenarios at once from // d_V_results; each scenario's own tripped branches are zeroed @@ -378,7 +520,7 @@ struct ScenarioSweepSession { RealVect get_residuals() const; // (n_scenarios,) real RealVect get_or_amps() const; // (n_scenarios * n_branches,) real RealVect get_ex_amps() const; // (n_scenarios * n_branches,) real - BatchTimings get_timings() const { return timings_; } + BatchTimings get_timings() const; // run() timings + cumulative adjoint counters // Per-scenario disconnected flag (1 == topology change islanded the grid, // scenario skipped/NaN; 0 == solved). Size n_scenarios(); empty before @@ -419,6 +561,14 @@ struct ScenarioSweepSession { // first (it references *base_state_). void _build_base_state(const std::vector& switchable_buses); + // Snapshot of the settings the live driver depends on (see + // ScenarioSweepDriverConfig). + ScenarioSweepDriverConfig _current_config() const; + + // Make an external stream's pending work visible to `cs` (event record + + // wait); 0 = nothing to wait for. + static void _wait_producer(std::uintptr_t producer_stream, void* cs); + // Derive, from gen_off_, the buses that lose every local controller per // row (row_pv_to_pq), the union of those (required, sorted, restricted to // buses that need a reserved Vm/Q pair), and the slack participants each diff --git a/src/_cpp/timing_utils.hpp b/src/_cpp/timing_utils.hpp index eacf23f..bad1298 100644 --- a/src/_cpp/timing_utils.hpp +++ b/src/_cpp/timing_utils.hpp @@ -385,6 +385,19 @@ struct BatchTimings { int n_refactorize = 0; // number of refactorize calls (n_chunks * nb_iter - 1) int n_disconnected = 0; // contingencies skipped (would disconnect the grid) + // --- batched adjoint (ScenarioSweepSession::solve_JT_batch, differentiable + // wrapper) -- CUMULATIVE over the life of the batch driver, all zero + // until the first backward pass; NOT part of any aggregate above (the + // adjoint is a separate call from run()). --- + double t_adjoint_build_ms = 0.; // Jᵀ skeleton/map + buffers + cuDSS ANALYSIS (first backward only) + TimingEntry t_adjoint_first_factorize; // single FACTORIZATION of Jᵀ (first backward only) + TimingEntry t_adjoint_refactorize; // REFACTORIZATION of Jᵀ (every later backward after a new forward) + TimingEntry t_adjoint_solve; // SOLVE with Jᵀ (every backward) + int adjoint_n_analysis = 0; // 0 or 1 per driver life + int adjoint_n_factorize = 0; // 0 or 1 per driver life + int adjoint_n_refactorize = 0; + int adjoint_n_solve = 0; + // Total wall-clock time for all chunks (excludes one-time setup). double t_chunks_total_wall_ms() const { return (t_tile_V + t_tile_Ybus + t_patch_Ybus diff --git a/src/gpusim2grid/__init__.py b/src/gpusim2grid/__init__.py index 6a535c9..c7d90ce 100644 --- a/src/gpusim2grid/__init__.py +++ b/src/gpusim2grid/__init__.py @@ -2,7 +2,7 @@ # License, v. 2.0. If a copy of the MPL was not distributed with this # file, You can obtain one at https://mozilla.org/MPL/2.0/. -__version__ = "0.1.2" +__version__ = "0.2.0-rc0" # Import lightsim2grid first so liblightsim2grid_core.so is loaded into the # process: the compiled _gpusim2grid extension links against it (the zero-copy diff --git a/src/gpusim2grid/_ls2g_utils.py b/src/gpusim2grid/_ls2g_utils.py index b7e01d5..80e2007 100644 --- a/src/gpusim2grid/_ls2g_utils.py +++ b/src/gpusim2grid/_ls2g_utils.py @@ -232,6 +232,16 @@ class InjectionElements: # in Sbus -- it is solved for). Mirrors lightsim2grid's SbusPolicy. gen_target_q_mvar: np.ndarray = None gen_vreg_on: np.ndarray = None + # (n_load,) AC-solver bus id of every load (-1 for a disconnected one) and + # the base-case per-element values the snapshot was taken at (MW / MVAr): + # what gpusim2grid.differentiable.BatchPowerFlow needs to build its own + # device-side element->bus map and to fill a ``None`` input with the + # grid's own values. Feeding these bases back through + # :func:`build_bus_injections` reproduces the base-case Sbus. + load_bus: np.ndarray = None + load_p_base: np.ndarray = None + load_q_base: np.ndarray = None + gen_p_base: np.ndarray = None def extract_injection_elements(grid, n_bus): @@ -321,13 +331,17 @@ def _scatter(bus, sel): gen_target_q = np.array([float(g.target_q_mvar) for g in gens], dtype=np.float64) gen_vreg_on = np.array([bool(g.voltage_regulator_on) for g in gens], dtype=bool) + load_bus_out = np.where(load_status, load_bus, -1) + return InjectionElements( n_load=n_load, n_gen=n_gen, n_bus=int(n_bus), sn_mva=float(grid.get_sn_mva()), load_sel=load_sel, gen_sel=gen_sel, scatter_load=scatter_load, scatter_gen=scatter_gen, const_mw=const_mw, gen_bus=gen_bus_out, - gen_target_q_mvar=gen_target_q, gen_vreg_on=gen_vreg_on) + gen_target_q_mvar=gen_target_q, gen_vreg_on=gen_vreg_on, + load_bus=load_bus_out, + load_p_base=load_p_base, load_q_base=load_q_base, gen_p_base=gen_p_base) def build_bus_injections(elements, load_p, load_q, gen_p, gen_off=None): diff --git a/src/gpusim2grid/differentiable/__init__.py b/src/gpusim2grid/differentiable/__init__.py index 0491d21..562a515 100644 --- a/src/gpusim2grid/differentiable/__init__.py +++ b/src/gpusim2grid/differentiable/__init__.py @@ -4,5 +4,6 @@ from ._power_flow_op import PowerFlowFunction, solve_power_flow from ._flows import compute_flows +from ._batch_pf import BatchPowerFlow -__all__ = ["PowerFlowFunction", "solve_power_flow", "compute_flows"] \ No newline at end of file +__all__ = ["PowerFlowFunction", "solve_power_flow", "compute_flows", "BatchPowerFlow"] diff --git a/src/gpusim2grid/differentiable/_batch_pf.py b/src/gpusim2grid/differentiable/_batch_pf.py new file mode 100644 index 0000000..550ce03 --- /dev/null +++ b/src/gpusim2grid/differentiable/_batch_pf.py @@ -0,0 +1,472 @@ +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at https://mozilla.org/MPL/2.0/. + +""" +BatchPowerFlow — batched, differentiable AC power flow for PyTorch, driven +by the same per-element inputs as lightsim2grid's ``ScenarioSweep``. + + pf = BatchPowerFlow.from_lsgrid(grid, nb_iter=6) + V = pf(load_p=..., load_q=..., gen_p=..., gen_v=..., + line_status=..., trafo_status=...) # complex (n_scen, n_bus) + +Every input is a ``(n_scenarios, n_elements)`` tensor (or ``None`` = the +grid's own base-case value in every row); row ``i`` is one independent +scenario: its own injections, its own voltage set-points and its own set of +disconnected branches (``line_status`` / ``trafo_status`` are boolean masks +with **True = connected**, grid2op's convention; a row trips the branches +whose mask is False). The whole batch is solved in ONE GPU pass by a +:class:`gpusim2grid.ScenarioSweepGPU` session that this object keeps alive, +so consecutive calls with the same number of rows reuse everything: no +cuDSS analysis, no first factorization, no re-upload of the grid, only the +new rows move to the device (see ``ScenarioSweepGPU.driver_build_counter``). + +Differentiable inputs: ``load_p``, ``load_q``, ``gen_p`` (through Sbus) and +``gen_v`` (through the seeded voltage magnitude). ``line_status`` / +``trafo_status`` are discrete and carry no gradient. + +Adjoint math (batched) +---------------------- +Each row ``s`` solves ``G(x_s; Sbus_s, Vm_s) = 0`` -- Newton-Raphson on the +augmented system lightsim2grid poses (see ``_power_flow_op.py``) -- and +returns ``V_s`` (a function of the unknowns ``x_s`` and, at Vm-fixed buses, +of the set-point directly). With the code Jacobian ``J_s = dS_calc/dx`` +(positive diagonal, see ``_power_flow_op.py`` for the sign discussion): + + x̄_s = projection of ḡV_s onto the theta / Vm unknowns + (theta_col_of_bus / vm_col_of_bus maps, same formulas as the + single-system op, vectorised over rows) + λ_s = J_sᵀ⁻¹ x̄_s (batched cuDSS solve of the + explicitly transposed system) + Sbus̄_s = +λ_s gathered by p_row_of_bus / q_row_of_bus (real / imag) + +then autograd maps Sbus̄ back to ``load_p`` / ``load_q`` / ``gen_p`` through +the (affine, natively differentiable) element→bus assembly done in torch. + +For ``gen_v``: a Vm-fixed bus ``k`` keeps ``|V_k| = gen_v`` throughout the +solve, so ``V_k = gen_v · e^{jθ_k*}`` with ``θ_k*`` (and every other unknown) +depending on ``gen_v`` through ``G``. Hence + + ḡ(gen_v)_k = Re( e^{-jθ_k} · ḡV_k ) (direct term, Python side) + - λ_s · ∂S_calc/∂Vm_k (indirect term, + gen_v_adjoint_kernel) + +where ``∂S_calc/∂Vm_k`` is the dS/dVm column ``fill_J`` never stores for a +Vm-fixed bus, evaluated on the row's own patched Ybus. The minus sign is the +implicit-function sign: ``dx/dVm_k = -J⁻¹ ∂S_calc/∂Vm_k``. Only generators +whose own bus is Vm-fixed (pv or slack) get a non-zero gradient -- exactly +the ones ``set_gen_v`` acts on; a NaN entry (= "keep the base-case voltage") +gets 0. + +Row bookkeeping: the session works in *active-slot* order (rows the +connectivity pre-check drops are compacted out); all of that stays in C++ +(``solve_JT_batch_dlpack`` takes and returns ORIGINAL row order). Dropped rows +are NaN in ``V`` and get a zero gradient; callers mask them out of the loss. + +Lazy Jᵀ: nothing adjoint-related is built until the first ``backward()``, +which creates the transposed pattern + J→Jᵀ position map, the buffers and a +second cuDSS batch context (one ANALYSIS + one FACTORIZATION). Every later +backward only permutes the values, REFACTORIZES (once per new forward) and +SOLVES. + +Jacobian lifetime: by default backward reads the converged Jacobians (and, +for ``gen_v``, the patched Ybus values) straight from the session's chunk +buffers, which the *next* forward overwrites. A ``run_counter`` guard turns a +forward→forward→backward(first) pattern into a clear error; pass +``snapshot_jacobian=True`` to clone those buffers in every forward instead +(one D2D copy of ``capacity x nnz_J`` reals and ``capacity x nnz_Y`` +complex, kept until backward). +""" + +import numpy as np +import torch +from torch import Tensor + +from .. import _gpusim2grid as _cpp +from .._ls2g_utils import extract_branch_data +from ..scenario_sweep.gpu_facade import ScenarioSweepGPU +from ._flows import compute_flows as _compute_flows_torch + + +__all__ = ["BatchPowerFlow"] + + +class BatchPowerFlow: + """Batched differentiable AC power flow (see the module docstring). + + Build it with :meth:`from_lsgrid`; call it like a function (or use + :meth:`forward`). + """ + + def __init__(self, sweep, *, snapshot_jacobian=False): + if sweep._elements is None: + raise ValueError( + "BatchPowerFlow needs a ScenarioSweepGPU built from a lightsim2grid " + "grid (explicit-array/tuple mode has no loads/generators to map).") + self._sweep = sweep + self._solver = sweep.solver # _ScenarioSweepSolver + self._solver.fixed_batch_capacity = True # always one chunk (adjoint) + self.snapshot_jacobian = bool(snapshot_jacobian) + + self._dev = torch.device("cuda", _device_index(sweep)) + self._rdtype = torch.float32 if bool(_cpp.is_fp32) else torch.float64 + self._cdtype = torch.complex64 if bool(_cpp.is_fp32) else torch.complex128 + + el = sweep._elements + dev, rdt = self._dev, self._rdtype + sn = float(el.sn_mva) + self.sn_mva = sn + self.n_bus = int(el.n_bus) + self.n_load = int(el.n_load) + self.n_gen = int(el.n_gen) + grid = sweep._grid + self.n_line = len(grid.get_lines()) + self.n_trafo = len(grid.get_trafos()) + self.n_branch = self.n_line + self.n_trafo + + # Element -> bus assembly (mirrors _ls2g_utils.build_bus_injections): + # Sbus_pu = const + Σ gen_p/sn − Σ (load_p + j load_q)/sn. + self._const_re = torch.as_tensor(np.ascontiguousarray(el.const_mw.real) / sn, dtype=rdt, device=dev) + self._const_im = torch.as_tensor(np.ascontiguousarray(el.const_mw.imag) / sn, dtype=rdt, device=dev) + self._gen_sel = torch.as_tensor(np.asarray(el.gen_sel, dtype=np.int64), device=dev) + self._gen_bus_sel = torch.as_tensor(np.asarray(el.gen_bus[el.gen_sel], dtype=np.int64), device=dev) + self._load_sel = torch.as_tensor(np.asarray(el.load_sel, dtype=np.int64), device=dev) + self._load_bus_sel = torch.as_tensor(np.asarray(el.load_bus[el.load_sel], dtype=np.int64), device=dev) + self._gen_bus_all = torch.as_tensor(np.asarray(el.gen_bus, dtype=np.int64), device=dev) + self._gen_bus_np = np.ascontiguousarray(el.gen_bus, dtype=np.int32) + self._is_vm_fixed = torch.as_tensor(self._solver.is_vm_fixed_bus, device=dev) + + # Base-case per-element values: what a ``None`` input means. + self._load_p_base = torch.tensor(np.array(el.load_p_base), dtype=rdt, device=dev) + self._load_q_base = torch.tensor(np.array(el.load_q_base), dtype=rdt, device=dev) + self._gen_p_base = torch.tensor(np.array(el.gen_p_base), dtype=rdt, device=dev) + + # Branch data for compute_flows() (lines-then-trafos, solver numbering). + (b_from, b_to, yff, yft, ytf, ytt, vn_kv, _sn), _, _ = extract_branch_data(grid) + self._branch_from = torch.as_tensor(np.asarray(b_from, dtype=np.int64), device=dev) + self._branch_to = torch.as_tensor(np.asarray(b_to, dtype=np.int64), device=dev) + self._yff = torch.as_tensor(np.asarray(yff), dtype=self._cdtype, device=dev) + self._yft = torch.as_tensor(np.asarray(yft), dtype=self._cdtype, device=dev) + self._ytf = torch.as_tensor(np.asarray(ytf), dtype=self._cdtype, device=dev) + self._ytt = torch.as_tensor(np.asarray(ytt), dtype=self._cdtype, device=dev) + self._bus_vn_kv = torch.as_tensor(np.asarray(vn_kv), dtype=rdt, device=dev) + + # Call-to-call state. + self._last_n_scen = None + self._topology_mask = None # (n_scen, n_branch) bool, True = tripped + self._topology_in_session = False + self._pending_topology = None # ragged list to hand to the session on the next run + self._gen_v_in_session = False + + # ------------------------------------------------------------------ build + @classmethod + def from_lsgrid(cls, grid, *, nb_iter=4, handle_disconnected_grid=False, + strategy="direct_refactor_every", snapshot_jacobian=False, + device=None, reordering_alg=None, matching_alg=None, + pivot_epsilon_alg=None, use_distributed_slack=True, + scaling_max_voltage_change=None, max_dVa=None, max_dVm=None, + init_from_n_powerflow=True, max_iter_base=10, tol_base=1e-8, + precision=None): + """Build from a *solved* lightsim2grid grid (``grid.ac_pf`` done). + + The keyword arguments are :class:`gpusim2grid.ScenarioSweepGPU`'s + (same meaning), plus ``strategy`` (linear-solve strategy string) and + ``snapshot_jacobian`` (see the module docstring). ``nb_iter`` is the + fixed Newton-Raphson iteration count per row: raise it (and lower + ``tol_base``) when gradients must be accurate -- the adjoint is exact + only at a converged solution. + """ + sweep = ScenarioSweepGPU( + grid, init_from_n_powerflow=init_from_n_powerflow, precision=precision, + nb_iter=nb_iter, max_iter_base=max_iter_base, tol_base=tol_base, + device=device, handle_disconnected_grid=handle_disconnected_grid, + reordering_alg=reordering_alg, matching_alg=matching_alg, + pivot_epsilon_alg=pivot_epsilon_alg, + scaling_max_voltage_change=scaling_max_voltage_change, + max_dVa=max_dVa, max_dVm=max_dVm, + use_distributed_slack=use_distributed_slack) + sweep.strategy = strategy + return cls(sweep, snapshot_jacobian=snapshot_jacobian) + + # --------------------------------------------------------------- forward + def __call__(self, *args, **kwargs): + return self.forward(*args, **kwargs) + + def forward(self, load_p=None, load_q=None, gen_p=None, gen_v=None, + line_status=None, trafo_status=None): + """Solve one scenario per row; returns complex ``V`` ``(n_scen, n_bus)`` + on the GPU (per-unit, AC-solver bus numbering; NaN rows = scenarios the + connectivity pre-check dropped). + + load_p, load_q : (n_scen, n_load) MW / MVAr (differentiable) + gen_p : (n_scen, n_gen) MW (differentiable) + gen_v : (n_scen, n_gen) vm_pu (differentiable; NaN + = keep the base-case voltage for that (row, gen)) + line_status : (n_scen, n_line) bool, True = connected + trafo_status : (n_scen, n_trafo) bool, True = connected + ``None`` = the grid's base-case value in every row. At least one + input must be given (it fixes n_scen). + """ + n_scen = self._infer_n_scen(load_p, load_q, gen_p, gen_v, line_status, trafo_status) + load_p = self._as_input(load_p, self.n_load, self._load_p_base, n_scen, "load_p") + load_q = self._as_input(load_q, self.n_load, self._load_q_base, n_scen, "load_q") + gen_p = self._as_input(gen_p, self.n_gen, self._gen_p_base, n_scen, "gen_p") + gen_v = None if gen_v is None else self._as_input(gen_v, self.n_gen, None, n_scen, "gen_v") + + self._apply_topology(line_status, trafo_status, n_scen) + + if n_scen != self._last_n_scen: + # Capacity == n_scen: one chunk, whatever rows get islanded. + self._solver.batch_size = n_scen + self._last_n_scen = n_scen + + # Element -> bus, in plain (differentiable) torch. + inv_sn = 1.0 / self.sn_mva + P = self._const_re.unsqueeze(0).expand(n_scen, -1).clone() + Q = self._const_im.unsqueeze(0).expand(n_scen, -1).clone() + if self._gen_sel.numel(): + P = P.index_add(1, self._gen_bus_sel, gen_p[:, self._gen_sel] * inv_sn) + if self._load_sel.numel(): + P = P.index_add(1, self._load_bus_sel, -load_p[:, self._load_sel] * inv_sn) + Q = Q.index_add(1, self._load_bus_sel, -load_q[:, self._load_sel] * inv_sn) + + return _BatchPowerFlowOp.apply(P, Q, gen_v, self) + + # --------------------------------------------------------------- results + def get_disconnected(self): + """(n_scen,) int: 1 where the last call dropped the row (NaN).""" + return self._sweep.get_disconnected() + + def last_residuals(self): + """(n_scen,) float: ``‖F‖∞`` of each row after the last call.""" + return self._solver.residuals.to_numpy() + + def converged(self, tol=1e-6): + """(n_scen,) bool: residual <= tol after the last call.""" + return self.last_residuals() <= tol + + @property + def timings(self): + """:class:`BatchTimings` of the last call (+ cumulative adjoint counters).""" + return self._solver.timings + + @property + def sweep(self): + """The underlying :class:`gpusim2grid.ScenarioSweepGPU` (escape hatch).""" + return self._sweep + + @property + def device(self): + return self._dev + + def compute_flows(self, V): + """Branch flows (``p_or_mw``, ``q_or_mvar``, ``p_ex_mw``, ``q_ex_mvar``, + ``i_or_a``, ``i_ex_a``) of shape ``(n_scen, n_branch)`` from ``V`` + ``(n_scen, n_bus)``; pure torch, differentiable. Branches a row tripped + are NOT zeroed here (mask them with the status inputs if needed).""" + return _compute_flows_torch(V, self._yff, self._yft, self._ytf, self._ytt, + self._branch_from, self._branch_to, + self._bus_vn_kv, self.sn_mva) + + # --------------------------------------------------------------- helpers + def _infer_n_scen(self, *inputs): + n = None + for x in inputs: + if x is None: + continue + shape = tuple(x.shape) if hasattr(x, "shape") else np.shape(x) + if len(shape) != 2: + raise ValueError( + f"every input must be 2-D (n_scenarios, n_elements); got shape {shape}") + if n is None: + n = int(shape[0]) + elif int(shape[0]) != n: + raise ValueError( + f"all inputs must share the same number of rows; got {n} and {shape[0]}") + if n is None: + raise ValueError( + "BatchPowerFlow needs at least one input (load_p, load_q, gen_p, gen_v, " + "line_status or trafo_status) to know the number of scenarios") + if n <= 0: + raise ValueError("the number of scenarios must be > 0") + return n + + def _as_input(self, x, n_cols, base, n_scen, name): + if x is None: + return base.unsqueeze(0).expand(n_scen, n_cols) + if isinstance(x, Tensor): + t = x.to(device=self._dev, dtype=self._rdtype) + else: + t = torch.as_tensor(np.asarray(x), dtype=self._rdtype, device=self._dev) + if t.shape != (n_scen, n_cols): + raise ValueError( + f"'{name}' must have shape ({n_scen}, {n_cols}), got {tuple(t.shape)}") + return t + + def _as_status(self, x, n_cols, n_scen, name): + if x is None: + return torch.ones(n_scen, n_cols, dtype=torch.bool, device=self._dev) + if isinstance(x, Tensor): + t = x.to(device=self._dev, dtype=torch.bool) + else: + t = torch.as_tensor(np.asarray(x, dtype=bool), device=self._dev) + if t.shape != (n_scen, n_cols): + raise ValueError( + f"'{name}' must have shape ({n_scen}, {n_cols}), got {tuple(t.shape)}") + return t + + def _apply_topology(self, line_status, trafo_status, n_scen): + """Decide what the session's topology must be for this call; the + ragged list (if any) is handed over by the op AFTER the injections + (the session checks the row counts against them).""" + self._pending_topology = None + if line_status is None and trafo_status is None: + # No trips wanted. Only touch the session if it still holds trips + # or its row count is stale. + if self._topology_mask is not None or ( + self._topology_in_session and n_scen != self._last_n_scen): + self._pending_topology = [[] for _ in range(n_scen)] + self._topology_mask = None + return + + tripped = torch.cat([ + ~self._as_status(line_status, self.n_line, n_scen, "line_status"), + ~self._as_status(trafo_status, self.n_trafo, n_scen, "trafo_status")], dim=1) + prev = self._topology_mask + if prev is not None and prev.shape == tripped.shape and torch.equal(prev, tripped): + return # unchanged: the session keeps its topology (hot path) + + # Dense mask -> ragged list of tripped branch ids (lines-then-trafos). + nz = tripped.nonzero() + if nz.numel(): + rows = nz[:, 0].cpu().numpy() + cols = nz[:, 1].cpu().numpy() + counts = np.bincount(rows, minlength=n_scen) + splits = np.split(cols, np.cumsum(counts)[:-1]) + ragged = [c.tolist() for c in splits] + else: + ragged = [[] for _ in range(n_scen)] + self._pending_topology = ragged + self._topology_mask = tripped.clone() + + +def _device_index(sweep): + """CUDA ordinal the session lives on (the facade normalises ``device``).""" + try: + return int(torch.from_dlpack(sweep.solver.v_base_dlpack()).device.index) + except Exception: # pragma: no cover - defensive + return 0 + + +class _BatchPowerFlowOp(torch.autograd.Function): + """(P, Q) per-unit bus injections (+ optional gen_v) -> V, with the + batched adjoint backward. Not public: :class:`BatchPowerFlow` builds + (P, Q) from the element inputs so autograd gets their gradients for free.""" + + @staticmethod + def forward(ctx, P: Tensor, Q: Tensor, gen_v, pf: BatchPowerFlow) -> Tensor: + solver = pf._solver + dev = pf._dev + stream = torch.cuda.current_stream(dev).cuda_stream + + S = torch.complex(P, Q).contiguous() + solver.set_injections_dlpack(S.__dlpack__(), stream) + if pf._pending_topology is not None: + pf._sweep.set_topology(pf._pending_topology) + pf._topology_in_session = True + pf._pending_topology = None + if gen_v is not None: + solver.set_gen_v_dlpack(gen_v.detach().contiguous().__dlpack__(), pf._gen_bus_np, stream) + pf._gen_v_in_session = True + elif pf._gen_v_in_session: + solver.clear_gen_v() + pf._gen_v_in_session = False + + want_gen_v = gen_v is not None and ctx.needs_input_grad[2] + needs_grad = ctx.needs_input_grad[0] or ctx.needs_input_grad[1] or want_gen_v + solver.keep_final_jacobian = bool(needs_grad) + solver.run() + + V = torch.from_dlpack(solver.v_results_dlpack()).clone() + + ctx.pf = pf + ctx.run_id = solver.run_counter + ctx.want_gen_v = bool(want_gen_v) + ctx.J = ctx.Y = None + if needs_grad and pf.snapshot_jacobian: + ctx.J = torch.from_dlpack(solver.j_values_dlpack()).clone() + if want_gen_v: + ctx.Y = torch.from_dlpack(solver.ybus_values_dlpack()).clone() + ctx.save_for_backward(V, gen_v) + return V + + @staticmethod + def backward(ctx, grad_V: Tensor): + pf = ctx.pf + solver = pf._solver + V, gen_v = ctx.saved_tensors + dev = V.device + rdtype = pf._rdtype + n_scen, n_bus = V.shape + + if ctx.J is None and solver.run_counter != ctx.run_id: + raise RuntimeError( + "BatchPowerFlow.backward: another forward() ran after the one this " + "gradient belongs to, overwriting its Jacobians on the GPU. Call " + "backward() before the next forward(), or build the model with " + "snapshot_jacobian=True to keep a copy per forward.") + + theta_col = torch.as_tensor(solver.theta_col_of_bus, device=dev) + vm_col = torch.as_tensor(solver.vm_col_of_bus, device=dev) + p_row = torch.as_tensor(solver.p_row_of_bus, device=dev) + q_row = torch.as_tensor(solver.q_row_of_bus, device=dev) + dim_J = int(solver.dim_J) + + # Cotangent projection (same formulas as _power_flow_op.py, per row). + # NaN buses (islanded rows / masked components) carry no cotangent. + valid = torch.isfinite(V.real) & torch.isfinite(V.imag) + gV = torch.where(valid, grad_V, torch.zeros_like(grad_V)) + Vs = torch.where(valid, V, torch.ones_like(V)) + Vn = Vs / Vs.abs() + proj_th = (Vs.conj() * gV).imag.to(rdtype) + proj_vm = (Vn.conj() * gV).real.to(rdtype) + + xbar = torch.zeros(n_scen, dim_J, dtype=rdtype, device=dev) + th_mask = theta_col >= 0 + vm_mask = vm_col >= 0 + xbar[:, theta_col[th_mask]] = proj_th[:, th_mask] + xbar[:, vm_col[vm_mask]] = proj_vm[:, vm_mask] + xbar = xbar.contiguous() + + stream = torch.cuda.current_stream(dev).cuda_stream + j_cap = y_cap = v_cap = None + if ctx.J is not None: + j_cap = ctx.J.__dlpack__() + v_cap = V.detach().contiguous().__dlpack__() + if ctx.Y is not None: + y_cap = ctx.Y.__dlpack__() + lam_cap, gvm_cap = solver.solve_JT_batch_dlpack( + xbar.__dlpack__(), j_cap, y_cap, v_cap, ctx.want_gen_v, stream) + lam = torch.from_dlpack(lam_cap).clone() + + # Sbus gradient: +λ (see _power_flow_op.py on the sign). + grad_P = torch.zeros(n_scen, n_bus, dtype=rdtype, device=dev) + grad_Q = torch.zeros(n_scen, n_bus, dtype=rdtype, device=dev) + p_mask = p_row >= 0 + q_mask = q_row >= 0 + grad_P[:, p_mask] = lam[:, p_row[p_mask]] + grad_Q[:, q_mask] = lam[:, q_row[q_mask]] + + grad_gen_v = None + if ctx.want_gen_v: + gvm = torch.from_dlpack(gvm_cap).clone() # indirect term (sign included) + g_bus = proj_vm + gvm # + direct term + gen_bus = pf._gen_bus_all + safe_bus = gen_bus.clamp(min=0) + ok = (gen_bus >= 0) & pf._is_vm_fixed[safe_bus] + g = g_bus[:, safe_bus] + g = torch.where(ok.unsqueeze(0), g, torch.zeros_like(g)) + g = torch.where(torch.isnan(gen_v), torch.zeros_like(g), g) + grad_gen_v = g + + return grad_P, grad_Q, grad_gen_v, None diff --git a/src/gpusim2grid/differentiable/_flows.py b/src/gpusim2grid/differentiable/_flows.py index 08aae43..d279205 100644 --- a/src/gpusim2grid/differentiable/_flows.py +++ b/src/gpusim2grid/differentiable/_flows.py @@ -14,7 +14,10 @@ base_A = sn_mva * 1e6 / (sqrt(3) * vn_kv[from] * 1e3) -All operations are natively differentiable via PyTorch autograd. +All operations are natively differentiable via PyTorch autograd. ``V`` may +carry any number of leading batch dimensions (``(..., n_bus)``, e.g. the +``(n_scen, n_bus)`` output of ``BatchPowerFlow``): the branch indexing is +done on the last axis and the branch arrays broadcast. The function body is framework-neutral: a future JAX variant only needs to swap the V indexing at the call site. """ @@ -25,7 +28,7 @@ def compute_flows( - V: Tensor, # complex [n_bus] on GPU + V: Tensor, # complex [..., n_bus] on GPU yff_eff: Tensor, # complex [n_branches] yft_eff: Tensor, # complex [n_branches] ytf_eff: Tensor, # complex [n_branches] @@ -44,10 +47,11 @@ def compute_flows( - ``p_ex_mw``, ``q_ex_mvar`` — MW/MVAr at the extremity terminal - ``i_or_a``, ``i_ex_a`` — ampere flows at each terminal - All values are real tensors of shape [n_branches]. + All values are real tensors of shape [..., n_branches] (the leading + dimensions of ``V``). """ - Vi = V[branch_from] # complex [n_branches] - Vj = V[branch_to] # complex [n_branches] + Vi = V[..., branch_from] # complex [..., n_branches] + Vj = V[..., branch_to] # complex [..., n_branches] I_or = yff_eff * Vi + yft_eff * Vj # origin terminal current I_ex = ytf_eff * Vi + ytt_eff * Vj # extremity terminal current @@ -64,4 +68,4 @@ def compute_flows( "q_ex_mvar": S_ex.imag * sn_mva, "i_or_a": I_or.abs() * base_A, "i_ex_a": I_ex.abs() * base_A, - } \ No newline at end of file + } diff --git a/src/gpusim2grid/scenario_sweep/__init__.py b/src/gpusim2grid/scenario_sweep/__init__.py index 05dd790..c7d1b30 100644 --- a/src/gpusim2grid/scenario_sweep/__init__.py +++ b/src/gpusim2grid/scenario_sweep/__init__.py @@ -548,11 +548,139 @@ def v_results_dlpack(self): """Zero-copy DLPack capsule of batch voltages, shape [n_scenarios, n_bus]. Requires ``run()`` to have been called. The capsule aliases live GPU - memory — calling ``run()`` again overwrites it in place. Clone before - a subsequent ``run()`` if a snapshot is needed. + memory — a later ``run()`` that reuses the batch driver (same + n_scenarios and settings) overwrites it in place, one that rebuilds + the driver frees it. Clone before a subsequent ``run()`` if a + snapshot is needed. """ return self._s.v_results_dlpack() + # --- driver persistence + differentiable path (see + # gpusim2grid.differentiable.BatchPowerFlow) --- + @property + def fixed_batch_capacity(self): + """bool: use ``batch_size`` verbatim as the driver's chunk capacity (no + rebalancing over the active count), so with ``batch_size >= + n_scenarios`` the batch is always one chunk. Default False.""" + return self._s.fixed_batch_capacity + + @fixed_batch_capacity.setter + def fixed_batch_capacity(self, value): + self._s.fixed_batch_capacity = bool(value) + + @property + def keep_final_jacobian(self): + """bool: refill the batched Jacobian at the converged voltages after + each run() so solve_JT_batch_dlpack() can use it. Default False.""" + return self._s.keep_final_jacobian + + @keep_final_jacobian.setter + def keep_final_jacobian(self, value): + self._s.keep_final_jacobian = bool(value) + + @property + def run_counter(self): + """int: number of run() calls so far.""" + return self._s.run_counter + + @property + def driver_build_counter(self): + """int: number of (cold) batch-driver builds -- allocation + cuDSS + ANALYSIS. Constant across run() calls that reuse the driver.""" + return self._s.driver_build_counter + + @property + def source_build_counter(self): + """int: number of batch-source builds (cold + warm runs). Constant + across hot runs (injections / gen_v only).""" + return self._s.source_build_counter + + @property + def capacity(self): + """int: the live driver's chunk capacity (0 before run()).""" + return self._s.capacity + + @property + def n_active(self): + """int: rows actually solved by the last run().""" + return self._s.n_active + + @property + def adjoint_ready(self): + """bool: the batched transposed system has been built (first backward).""" + return self._s.adjoint_ready + + @property + def nnz_J(self): + return self._s.nnz_J + + @property + def nnz_Y(self): + return self._s.nnz_Y + + @property + def p_row_of_bus(self): + return np.asarray(self._s.p_row_of_bus, dtype=np.int64) + + @property + def q_row_of_bus(self): + return np.asarray(self._s.q_row_of_bus, dtype=np.int64) + + @property + def theta_col_of_bus(self): + return np.asarray(self._s.theta_col_of_bus, dtype=np.int64) + + @property + def vm_col_of_bus(self): + return np.asarray(self._s.vm_col_of_bus, dtype=np.int64) + + @property + def is_vm_fixed_bus(self): + """(n_bus,) bool: |V| fixed at that bus (pv or slack).""" + return np.asarray(self._s.is_vm_fixed_bus, dtype=bool) + + def get_active_to_orig(self): + """(n_active,) int64: original scenario index of each active slot.""" + return np.asarray(self._s.get_active_to_orig(), dtype=np.int64) + + def j_skeleton(self): + """(outer, inner) int32 CSR structure of one Jacobian.""" + outer, inner = self._s.j_skeleton() + return np.asarray(outer, dtype=np.int32), np.asarray(inner, dtype=np.int32) + + def clear_gen_v(self): + """Drop any set_gen_v() override (rows keep the base-case voltage).""" + self._s.clear_gen_v() + + def set_injections_dlpack(self, capsule, producer_stream=0): + """Device path of set_injections(): a DLPack capsule of a + (n_scenarios, n_bus) contiguous complex PER-UNIT Sbus tensor on this + session's device (this build's precision). Consumes the capsule.""" + self._s.set_injections_dlpack(capsule, int(producer_stream)) + + def set_gen_v_dlpack(self, capsule, gen_bus, producer_stream=0): + """Device path of set_gen_v(gen_v, gen_bus): (n_scenarios, n_gen) + contiguous real tensor of vm_pu on this device. Consumes the capsule.""" + b = np.ascontiguousarray(gen_bus, dtype=np.int32) + self._s.set_gen_v_dlpack(capsule, b, int(producer_stream)) + + def solve_JT_batch_dlpack(self, rhs, j_values=None, ybus_values=None, v=None, + want_gen_v_grad=False, producer_stream=0): + """Batched adjoint solve; see ScenarioSweepSession.solve_JT_batch_dlpack. + Returns (lambda capsule, gvm capsule or None); clone both.""" + return self._s.solve_JT_batch_dlpack(rhs, j_values, ybus_values, v, + bool(want_gen_v_grad), int(producer_stream)) + + def j_values_dlpack(self): + """(capacity, nnz_J) real DLPack capsule aliasing the last chunk's + Jacobian values (active-slot order).""" + return self._s.j_values_dlpack() + + def ybus_values_dlpack(self): + """(capacity, nnz_Y) complex DLPack capsule aliasing the last chunk's + patched Ybus values (active-slot order).""" + return self._s.ybus_values_dlpack() + @property def n_scenarios(self): return self._s.n_scenarios diff --git a/src/gpusim2grid/scenario_sweep/gpu_facade.py b/src/gpusim2grid/scenario_sweep/gpu_facade.py index 771ae34..a67e944 100644 --- a/src/gpusim2grid/scenario_sweep/gpu_facade.py +++ b/src/gpusim2grid/scenario_sweep/gpu_facade.py @@ -612,13 +612,48 @@ def strategy(self): def strategy(self, value): self._inner.strategy = value + @property + def nb_iter(self): + """int: fixed NR iterations per scenario. Takes effect on the next + compute() without rebuilding the batch driver.""" + return self._inner.nb_iter + + @nb_iter.setter + def nb_iter(self, value): + self._inner.nb_iter = int(value) + + @property + def run_counter(self): + """int: number of compute() calls so far.""" + return self._inner.run_counter + + @property + def driver_build_counter(self): + """int: number of cold batch-driver builds (allocation + cuDSS + ANALYSIS). compute() reuses the driver whenever n_scenarios and the + settings are unchanged: only the injections (and, when the topology + changed, the patch arrays) move to the GPU, and the Jacobians are + REFACTORIZED rather than analysed again.""" + return self._inner.driver_build_counter + + @property + def source_build_counter(self): + """int: number of batch-source builds (cold + topology changes).""" + return self._inner.source_build_counter + + @property + def active_to_orig(self): + """(n_active,) int64: original row index of each solved batch slot.""" + return self._inner.get_active_to_orig() + @property def reordering_alg(self): """cuDSS CUDSS_CONFIG_REORDERING_ALG choice (str). Takes effect on the - next compute() (which always reruns cuDSS ANALYSIS). One of 'default' - (default), 'amd', 'nested_dissection', 'none'. 'btf_colamd'/'colamd' - are rejected by cuDSS (CUDSS_STATUS_NOT_SUPPORTED) in this class's - uniform-batch mode -- they only work on AcPfGPU's single-system solve.""" + next compute(), which rebuilds the batch driver (a new cuDSS ANALYSIS) + when it changed. One of 'default' (default), 'amd', + 'nested_dissection', 'none'. 'btf_colamd'/'colamd' are rejected by + cuDSS (CUDSS_STATUS_NOT_SUPPORTED) in this class's uniform-batch mode + -- they only work on AcPfGPU's single-system solve.""" return self._inner.reordering_alg @reordering_alg.setter @@ -628,9 +663,10 @@ def reordering_alg(self, value): @property def matching_alg(self): """cuDSS CUDSS_CONFIG_MATCHING_ALG choice (str). Takes effect on the - next compute() (which always reruns cuDSS ANALYSIS). 'none' (default) - is the only value cuDSS accepts in this class's uniform-batch mode -- - every other value raises RuntimeError (CUDSS_STATUS_NOT_SUPPORTED).""" + next compute(), which rebuilds the batch driver when it changed. + 'none' (default) is the only value cuDSS accepts in this class's + uniform-batch mode -- every other value raises RuntimeError + (CUDSS_STATUS_NOT_SUPPORTED).""" return self._inner.matching_alg @matching_alg.setter @@ -640,8 +676,8 @@ def matching_alg(self, value): @property def pivot_epsilon_alg(self): """cuDSS CUDSS_CONFIG_PIVOT_EPSILON_ALG choice (str). Takes effect on - the next compute() (which always reruns cuDSS ANALYSIS). One of - 'default' (default), 'scaled', 'static'.""" + the next compute(), which rebuilds the batch driver when it changed. + One of 'default' (default), 'scaled', 'static'.""" return self._inner.pivot_epsilon_alg @pivot_epsilon_alg.setter diff --git a/tests/python/test_batch_power_flow.py b/tests/python/test_batch_power_flow.py new file mode 100644 index 0000000..4ce60cc --- /dev/null +++ b/tests/python/test_batch_power_flow.py @@ -0,0 +1,507 @@ +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at https://mozilla.org/MPL/2.0/. + +"""gpusim2grid.differentiable.BatchPowerFlow: batched, differentiable AC power +flow driven by lightsim2grid-style per-element inputs. + +Forward is checked against ScenarioSweepGPU (the numpy path of the same +session class); every gradient is checked against FINITE DIFFERENCES -- +torch.autograd.gradcheck plus explicit central differences on a scalar loss +(FP64 build only: FP32 finite differences are too noisy). The reuse contract +(no cuDSS analysis / first factorization on repeated calls; the transposed +system built lazily on the first backward and only refactorized afterwards) is +asserted on the session counters. +""" +import warnings + +import numpy as np +import pytest + +from conftest import requires_gpu +from gpusim2grid._gpusim2grid import have_ls2g_bridge, is_fp32 +from test_handle_disconnected_grid import _solved_spur_grid + +pytestmark = requires_gpu + +torch = pytest.importorskip("torch", reason="PyTorch not installed -- skipping") + +needs_bridge = pytest.mark.skipif( + not have_ls2g_bridge, reason="needs the lightsim2grid C++ bridge") +fp64_only = pytest.mark.skipif( + bool(is_fp32), reason="finite-difference gradient checks need the FP64 build") + +NB_ITER = 12 +TOL = 1e-10 +RDT = torch.float32 if is_fp32 else torch.float64 + + +def _pf(grid, **kw): + from gpusim2grid.differentiable import BatchPowerFlow + kw.setdefault("nb_iter", NB_ITER) + kw.setdefault("tol_base", TOL) + return BatchPowerFlow.from_lsgrid(grid, **kw) + + +def _base_inputs(pf, n_scen, scales=None): + scales = np.ones(n_scen) if scales is None else np.asarray(scales, dtype=np.float64) + s = torch.tensor(scales, dtype=RDT, device="cuda")[:, None] + load_p = (pf._load_p_base[None, :] * s).clone() + load_q = (pf._load_q_base[None, :] * s).clone() + gen_p = pf._gen_p_base[None, :].expand(n_scen, -1).clone() + return load_p, load_q, gen_p + + +def _all_connected(pf, n_scen): + return (torch.ones(n_scen, pf.n_line, dtype=torch.bool, device="cuda"), + torch.ones(n_scen, pf.n_trafo, dtype=torch.bool, device="cuda")) + + +def _central_diff(f, x, idx, eps=1e-6): + xp = x.detach().clone(); xp[idx] += eps + xm = x.detach().clone(); xm[idx] -= eps + return (f(xp) - f(xm)) / (2 * eps) + + +# --------------------------------------------------------------------------- +# Forward +# --------------------------------------------------------------------------- + +class TestForward: + def test_matches_scenario_sweep_mixed_batch(self, ieee14_base_case, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + pf = _pf(grid) + n = 4 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1, 0.9, 1.05]) + line_status, trafo_status = _all_connected(pf, n) + line_status[1, 3] = False # row 1: trip line 3 + trafo_status[2, 0] = False # row 2: trip trafo 0 + gen_v = torch.full((n, pf.n_gen), float("nan"), dtype=RDT, device="cuda") + gen_v[3, 1] = 1.03 # row 3: a PV set-point change + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=gen_v, + line_status=line_status, trafo_status=trafo_status) + assert V.shape == (n, pf.n_bus) and V.is_cuda + assert np.all(pf.last_residuals() < 1e-6) + + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(load_p.cpu().numpy(), load_q.cpu().numpy(), + gen_p.cpu().numpy()) + sw.set_topology([[], [3], [pf.n_line + 0], []]) + sw.set_gen_v(gen_v.cpu().numpy()) + Vref = torch.from_dlpack(sw.compute(batch_size=n)).cpu().numpy() + np.testing.assert_allclose(V.cpu().numpy(), Vref, atol=solver_atol) + # Each input differs from the others' rows (the batch is really mixed). + assert not np.allclose(Vref[0], Vref[1]) and not np.allclose(Vref[0], Vref[2]) + + def test_none_inputs_reproduce_the_base_case(self, ieee14_base_case, solver_atol): + grid = ieee14_base_case["grid"] + pf = _pf(grid) + V = pf(gen_p=pf._gen_p_base[None, :].expand(3, -1)) + V_n = np.asarray(grid.get_V_solver()) + for r in range(3): + np.testing.assert_allclose(V[r].cpu().numpy(), V_n, atol=solver_atol) + with pytest.raises(ValueError, match="at least one input"): + pf() + + def test_status_false_means_tripped(self, ieee14_base_case, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + pf = _pf(grid) + line_status, _ = _all_connected(pf, 2) + line_status[0, 5] = False + V = pf(line_status=line_status) + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + lp, lq, gp = (a.cpu().numpy() for a in _base_inputs(pf, 2)) + sw.set_injections_from_elements(lp, lq, gp) + sw.set_topology([[5], []]) + Vref = torch.from_dlpack(sw.compute(batch_size=2)).cpu().numpy() + np.testing.assert_allclose(V.cpu().numpy(), Vref, atol=solver_atol) + assert not np.allclose(Vref[0], Vref[1]) + + @needs_bridge + def test_islanded_row_is_nan(self, solver_atol): + grid, _, spur_line, _ = _solved_spur_grid(distributed_slack=False) + pf = _pf(grid) + line_status, _ = _all_connected(pf, 3) + line_status[1, int(spur_line)] = False + V = pf(line_status=line_status) + assert np.all(np.isnan(V[1].cpu().numpy())) + assert np.all(np.isfinite(V[[0, 2]].cpu().numpy())) + assert list(pf.get_disconnected()) == [0, 1, 0] + assert pf.timings.n_chunks == 1 + + def test_compute_flows_batched(self, ieee14_base_case): + from gpusim2grid.differentiable import compute_flows + grid = ieee14_base_case["grid"] + pf = _pf(grid) + V = pf(gen_p=pf._gen_p_base[None, :].expand(2, -1)) + flows = pf.compute_flows(V) + assert flows["p_or_mw"].shape == (2, pf.n_branch) + single = compute_flows(V[0], pf._yff, pf._yft, pf._ytf, pf._ytt, + pf._branch_from, pf._branch_to, pf._bus_vn_kv, pf.sn_mva) + torch.testing.assert_close(flows["i_or_a"][0], single["i_or_a"]) + + +# --------------------------------------------------------------------------- +# Reuse contract +# --------------------------------------------------------------------------- + +class TestReuse: + def test_forward_reuse_and_lazy_adjoint(self, ieee14_base_case): + grid = ieee14_base_case["grid"] + pf = _pf(grid) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1, 0.9]) + line_status, trafo_status = _all_connected(pf, n) + line_status[1, 3] = False + + # forward only: the driver is built, nothing adjoint-related exists + V = pf(load_p=load_p, gen_p=gen_p, line_status=line_status, trafo_status=trafo_status) + t = pf.timings + assert pf.sweep.driver_build_counter == 1 and pf.sweep.source_build_counter == 1 + assert t.t_analysis_ms > 0 and t.n_refactorize == NB_ITER - 1 + assert t.adjoint_n_analysis == 0 and not pf.sweep.solver.adjoint_ready + + # same shape, new injections -> hot path + load_p2 = (load_p * 1.02).requires_grad_(True) + V2 = pf(load_p=load_p2, gen_p=gen_p, line_status=line_status, trafo_status=trafo_status) + t = pf.timings + assert pf.sweep.driver_build_counter == 1 and pf.sweep.source_build_counter == 1 + assert t.t_analysis_ms == 0.0 and t.t_alloc_ms == 0.0 and t.t_preprocess_ms == 0.0 + assert t.t_first_factorize.wall_ms == 0.0 and t.n_refactorize == NB_ITER + assert t.adjoint_n_analysis == 0 + + # first backward: J^T built once (analysis + factorization), solved once + V2.real.sum().backward() + t = pf.timings + assert pf.sweep.solver.adjoint_ready + assert (t.adjoint_n_analysis, t.adjoint_n_factorize, + t.adjoint_n_refactorize, t.adjoint_n_solve) == (1, 1, 0, 1) + assert t.t_adjoint_build_ms > 0 and t.t_adjoint_first_factorize.wall_ms > 0 + + # second backward on the SAME forward: solve only + V2 = pf(load_p=load_p2, gen_p=gen_p, line_status=line_status, trafo_status=trafo_status) + loss = V2.real.sum() + loss.backward(retain_graph=True) + loss.backward() + t = pf.timings + assert (t.adjoint_n_analysis, t.adjoint_n_factorize, + t.adjoint_n_refactorize, t.adjoint_n_solve) == (1, 1, 1, 3) + + # new topology (warm) + backward: refactorize only + line_status[2, 7] = False + V3 = pf(load_p=load_p2, gen_p=gen_p, line_status=line_status, trafo_status=trafo_status) + assert pf.sweep.driver_build_counter == 1 and pf.sweep.source_build_counter == 2 + assert pf.timings.t_analysis_ms == 0.0 + V3.real.sum().backward() + t = pf.timings + assert (t.adjoint_n_analysis, t.adjoint_n_factorize, t.adjoint_n_refactorize) == (1, 1, 2) + + # new n_scen: cold rebuild, the adjoint is rebuilt lazily on the next backward + lp4 = _base_inputs(pf, 5)[0].requires_grad_(True) + V4 = pf(load_p=lp4) + assert pf.sweep.driver_build_counter == 2 + assert pf.timings.adjoint_n_analysis == 0 + V4.real.sum().backward() + assert pf.timings.adjoint_n_analysis == 1 + + def test_backward_after_another_forward_raises(self, ieee14_base_case): + grid = ieee14_base_case["grid"] + pf = _pf(grid) + load_p, _, gen_p = _base_inputs(pf, 2, [1.0, 1.1]) + lp = load_p.clone().requires_grad_(True) + V1 = pf(load_p=lp, gen_p=gen_p) + pf(load_p=load_p * 1.01, gen_p=gen_p) + with pytest.raises(RuntimeError, match="snapshot_jacobian"): + V1.real.sum().backward() + + def test_snapshot_jacobian_allows_two_forwards(self, ieee14_base_case, solver_atol): + grid = ieee14_base_case["grid"] + pf = _pf(grid, snapshot_jacobian=True) + pf_ref = _pf(grid) + load_p, _, gen_p = _base_inputs(pf, 2, [1.0, 1.1]) + gen_v = torch.full((2, pf.n_gen), 1.02, dtype=RDT, device="cuda") + + lp = load_p.clone().requires_grad_(True) + gv = gen_v.clone().requires_grad_(True) + V1 = pf(load_p=lp, gen_p=gen_p, gen_v=gv) + pf(load_p=load_p * 1.3, gen_p=gen_p) # overwrites the chunk buffers + (V1.abs() ** 2).sum().backward() # ... but the snapshot survives + + lp_ref = load_p.clone().requires_grad_(True) + gv_ref = gen_v.clone().requires_grad_(True) + (pf_ref(load_p=lp_ref, gen_p=gen_p, gen_v=gv_ref).abs() ** 2).sum().backward() + torch.testing.assert_close(lp.grad, lp_ref.grad, atol=10 * solver_atol, rtol=1e-6) + torch.testing.assert_close(gv.grad, gv_ref.grad, atol=10 * solver_atol, rtol=1e-6) + + def test_non_default_stream(self, ieee14_base_case, solver_atol): + grid = ieee14_base_case["grid"] + pf = _pf(grid) + load_p, _, gen_p = _base_inputs(pf, 3, [1.0, 1.1, 0.9]) + + lp = load_p.clone().requires_grad_(True) + V = pf(load_p=lp, gen_p=gen_p) + V.abs().sum().backward() + + s = torch.cuda.Stream() + with torch.cuda.stream(s): + lp_s = (load_p.clone() * 1.0).requires_grad_(True) + V_s = pf(load_p=lp_s, gen_p=gen_p) + V_s.abs().sum().backward() + torch.cuda.synchronize() + torch.testing.assert_close(V_s, V, atol=solver_atol, rtol=0) + torch.testing.assert_close(lp_s.grad, lp.grad, atol=10 * solver_atol, rtol=1e-6) + + +# --------------------------------------------------------------------------- +# Jacobian / adjoint primitives +# --------------------------------------------------------------------------- + +class TestAdjointPrimitives: + @needs_bridge + @fp64_only + def test_converged_jacobian_matches_single_system(self, ieee14_base_case): + """Row 0 (base injections, no trip) of the kept batched J must equal the + single-system converged J of AcPfNrSession on the same augmented ledger.""" + from scipy.sparse import csr_matrix + from gpusim2grid._gpusim2grid import _make_acpf_session_from_lsgrid + grid = ieee14_base_case["grid"] + pf = _pf(grid) + load_p, _, gen_p = _base_inputs(pf, 2, [1.0, 1.15]) + lp = load_p.clone().requires_grad_(True) + pf(load_p=lp, gen_p=gen_p) # requires_grad -> keep_final_jacobian + sol = pf.sweep.solver + outer, inner = sol.j_skeleton() + J_all = torch.from_dlpack(sol.j_values_dlpack()).cpu().numpy() + assert J_all.shape == (2, sol.nnz_J) + n = sol.dim_J + J0 = csr_matrix((J_all[0], inner, outer), shape=(n, n)).toarray() + + ref = _make_acpf_session_from_lsgrid(grid, 30, 1e-10) + o_r, i_r, v_r = ref.get_J() + J_ref = csr_matrix((np.asarray(v_r), np.asarray(i_r), np.asarray(o_r)), + shape=(ref.dim_J, ref.dim_J)).toarray() + assert J_ref.shape == J0.shape + np.testing.assert_allclose(J0, J_ref, atol=1e-8, rtol=1e-8) + + @needs_bridge + @fp64_only + def test_solve_JT_batch_matches_spsolve(self): + from scipy.sparse import csr_matrix + from scipy.sparse.linalg import spsolve + grid, _, spur_line, _ = _solved_spur_grid(distributed_slack=False) + pf = _pf(grid) + n = 4 + load_p, _, gen_p = _base_inputs(pf, n, [1.0, 1.1, 0.9, 1.05]) + line_status, _ = _all_connected(pf, n) + line_status[1, 3] = False + line_status[2, int(spur_line)] = False # islanded row + lp = load_p.clone().requires_grad_(True) + pf(load_p=lp, gen_p=gen_p, line_status=line_status) + sol = pf.sweep.solver + dim_J = sol.dim_J + rng = np.random.default_rng(0) + rhs = rng.standard_normal((n, dim_J)) + rhs[0, 3] = np.nan # non-finite entries count as 0 + lam_cap, gvm = sol.solve_JT_batch_dlpack( + torch.tensor(rhs, device="cuda").__dlpack__()) + assert gvm is None + lam = torch.from_dlpack(lam_cap).cpu().numpy() + outer, inner = sol.j_skeleton() + J_all = torch.from_dlpack(sol.j_values_dlpack()).cpu().numpy() + a2o = sol.get_active_to_orig() + assert list(a2o) == [0, 1, 3] + rhs_clean = np.nan_to_num(rhs) + for slot, r in enumerate(a2o): + J = csr_matrix((J_all[slot], inner, outer), shape=(dim_J, dim_J)) + expected = spsolve(J.T.tocsr(), rhs_clean[r]) + np.testing.assert_allclose(lam[r], expected, atol=1e-8, rtol=1e-8) + assert np.all(lam[2] == 0.0) # dropped row + + +# --------------------------------------------------------------------------- +# Gradients vs finite differences +# --------------------------------------------------------------------------- + +@fp64_only +class TestGradients: + def test_gradcheck_ieee14_mixed_topology(self, ieee14_base_case): + grid = ieee14_base_case["grid"] + # gradcheck interleaves many forwards before it back-propagates the + # first output: that needs the per-forward Jacobian snapshots. + pf = _pf(grid, nb_iter=15, snapshot_jacobian=True) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1, 0.9]) + line_status, trafo_status = _all_connected(pf, n) + line_status[1, 3] = False + gen_v = torch.full((n, pf.n_gen), float("nan"), dtype=RDT, device="cuda") + gen_v[:, 1] = 1.04 + gen_v[2, 2] = 1.0 + inputs = tuple(x.clone().requires_grad_(True) for x in (load_p, load_q, gen_p, gen_v)) + + def f_real(lp, lq, gp, gv): + return pf(load_p=lp, load_q=lq, gen_p=gp, gen_v=gv, + line_status=line_status, trafo_status=trafo_status).real.sum() + + def f_abs(lp, lq, gp, gv): + return pf(load_p=lp, load_q=lq, gen_p=gp, gen_v=gv, + line_status=line_status, trafo_status=trafo_status).abs().sum() + + assert torch.autograd.gradcheck(f_real, inputs, eps=1e-5, atol=1e-3, rtol=1e-2, + nondet_tol=1e-7) + assert torch.autograd.gradcheck(f_abs, inputs, eps=1e-5, atol=1e-3, rtol=1e-2, + nondet_tol=1e-7) + + @needs_bridge + def test_gradcheck_with_islanded_row(self): + grid, _, spur_line, _ = _solved_spur_grid(distributed_slack=False) + pf = _pf(grid, nb_iter=15, snapshot_jacobian=True) + n = 3 + load_p, _, gen_p = _base_inputs(pf, n, [1.0, 1.05, 0.95]) + line_status, _ = _all_connected(pf, n) + line_status[1, int(spur_line)] = False + lp = load_p.clone().requires_grad_(True) + gp = gen_p.clone().requires_grad_(True) + + def f(lp, gp): + V = pf(load_p=lp, gen_p=gp, line_status=line_status) + valid = torch.isfinite(V.real) + return torch.where(valid, V, torch.zeros_like(V)).real.sum() + + assert torch.autograd.gradcheck(f, (lp, gp), eps=1e-5, atol=1e-3, rtol=1e-2, + nondet_tol=1e-7) + # the islanded row gets exactly zero gradient + f(lp, gp).backward() + assert torch.all(lp.grad[1] == 0) and torch.all(gp.grad[1] == 0) + assert torch.any(lp.grad[0] != 0) and torch.any(lp.grad[2] != 0) + + def _fd_check(self, pf, load_p, load_q, gen_p, gen_v, line_status, coords, tol=1e-4): + """Central differences of a weighted |V|^2 loss vs the analytic gradient.""" + n_bus = pf.n_bus + w = torch.linspace(0.5, 1.5, n_bus, dtype=RDT, device="cuda") + + def loss_of(lp, lq, gp, gv): + V = pf(load_p=lp, load_q=lq, gen_p=gp, gen_v=gv, line_status=line_status) + valid = torch.isfinite(V.real) + Vs = torch.where(valid, V, torch.zeros_like(V)) + return ((Vs.abs() ** 2) * w).sum() + + ins = {"load_p": load_p, "load_q": load_q, "gen_p": gen_p, "gen_v": gen_v} + grads = {k: v.clone().requires_grad_(True) for k, v in ins.items() if v is not None} + args = {k: grads.get(k) for k in ins} + loss_of(args["load_p"], args["load_q"], args["gen_p"], args["gen_v"]).backward() + for name, idx in coords: + def f(x, name=name): + a = {k: (v.detach() if v is not None else None) for k, v in args.items()} + a[name] = x + with torch.no_grad(): + return loss_of(a["load_p"], a["load_q"], a["gen_p"], a["gen_v"]).item() + fd = _central_diff(f, grads[name], idx) + an = grads[name].grad[idx].item() + assert abs(an - fd) <= tol * max(1.0, abs(fd)), (name, idx, an, fd) + return grads + + def test_central_differences_ieee14_distributed_slack(self, ieee14_base_case): + grid = ieee14_base_case["grid"] + pf = _pf(grid, nb_iter=15) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1, 0.9]) + line_status, _ = _all_connected(pf, n) + line_status[2, 3] = False + gen_v = torch.tensor([[1.06, 1.045, 1.01, 1.07, 1.09]] * n, dtype=RDT, device="cuda") + gen_v[1, 3] = float("nan") + coords = [("load_p", (0, 2)), ("load_p", (2, 5)), ("load_q", (1, 4)), + ("gen_p", (0, 1)), ("gen_p", (2, 3)), + ("gen_v", (0, 0)), # slack generator: Vm-fixed too + ("gen_v", (1, 1)), ("gen_v", (2, 2)), ("gen_v", (2, 4)), + ("gen_v", (1, 3))] # NaN entry -> gradient must be 0 + grads = self._fd_check(pf, load_p, load_q, gen_p, gen_v, line_status, coords) + assert grads["gen_v"].grad[1, 3] == 0.0 + assert torch.all(grads["gen_v"].grad[0] != 0) + + @needs_bridge + def test_central_differences_hvdc_droop_and_remote_vc(self): + from test_augmented_features import _solved_hvdc_droop_grid + from test_diff_augmented import _solved_remote_gen_grid + for grid in (_solved_hvdc_droop_grid(), _solved_remote_gen_grid()): + pf = _pf(grid, nb_iter=20) + n = 2 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.05]) + line_status, _ = _all_connected(pf, n) + line_status[1, 2] = False + gen_v = torch.full((n, pf.n_gen), float("nan"), dtype=RDT, device="cuda") + gen_v[:, 1] = 1.045 + coords = [("load_p", (0, 3)), ("load_q", (1, 1)), ("gen_p", (1, 2)), + ("gen_v", (0, 1)), ("gen_v", (1, 1))] + self._fd_check(pf, load_p, load_q, gen_p, gen_v, line_status, coords) + + def test_gen_v_disconnected_and_colocated(self, solver_atol): + """A disconnected generator's gen_v is ignored (zero gradient); two + generators on the same bus report the same (bus) gradient.""" + pp = pytest.importorskip("pandapower") + import pandapower.networks as pn + from lightsim2grid.network import init_from_pandapower + from lightsim2grid.lightsim2grid_cpp import AlgorithmType + with warnings.catch_warnings(): + warnings.simplefilter("ignore") + net = pn.case14() + pp.create_gen(net, bus=int(net.gen.bus.iloc[1]), p_mw=5.0, + vm_pu=float(net.gen.vm_pu.iloc[1])) # gen 4: co-located with gen 1 + pp.runpp(net) + grid = init_from_pandapower(net) # gens: pp gens 0..4, then the ext_grid + grid.deactivate_gen(3) # disconnected gen (bus 7 turns PQ) + grid.tell_solver_need_reset() + grid.change_algorithm(AlgorithmType.NR_KLU) + n_model = grid.get_bus_vn_kv().shape[0] + v0 = grid.dc_pf(np.ones(n_model, dtype=complex), 1, 1e-6) + grid.ac_pf(v0.copy(), 30, 1e-10) + pf = _pf(grid, nb_iter=15) + assert pf.n_gen == 6 and int(pf._gen_bus_all[3]) == -1 + assert int(pf._gen_bus_all[4]) == int(pf._gen_bus_all[1]) + n = 2 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1]) + gen_v = torch.full((n, pf.n_gen), float("nan"), dtype=RDT, device="cuda") + gen_v[:, 1] = 1.05 + gen_v[:, 4] = 1.05 # the co-located generator: same value + gen_v[:, 3] = 1.2 # disconnected: must be ignored + V_ignore = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, + gen_v=torch.where(torch.arange(pf.n_gen, device="cuda") == 3, + torch.full_like(gen_v, float("nan")), gen_v)) + gv = gen_v.clone().requires_grad_(True) + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=gv) + torch.testing.assert_close(V, V_ignore, atol=solver_atol, rtol=0) + (V.abs() ** 2).sum().backward() + assert torch.all(gv.grad[:, 3] == 0) + torch.testing.assert_close(gv.grad[:, 1], gv.grad[:, 4]) + assert torch.all(gv.grad[:, 1] != 0) + + @needs_bridge + def test_matches_single_system_adjoint_on_untripped_row(self, ieee14_base_case): + """Row 0 (no trip): the batched Sbus gradient equals the single-system + PowerFlowFunction one on the same augmented system.""" + from gpusim2grid.differentiable import solve_power_flow + grid = ieee14_base_case["grid"] + pf = _pf(grid, nb_iter=15) + n = 2 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1]) + line_status, _ = _all_connected(pf, n) + line_status[1, 3] = False + lp = load_p.clone().requires_grad_(True) + lq = load_q.clone().requires_grad_(True) + V = pf(load_p=lp, load_q=lq, gen_p=gen_p, line_status=line_status) + V[0].real.sum().backward() + + # Single system at the same Sbus: gradient w.r.t. Sbus, mapped to loads. + Sbus0 = np.asarray(grid.get_Sbus_solver()) + Sr = torch.tensor(Sbus0.real, dtype=torch.float64, device="cuda", requires_grad=True) + Si = torch.tensor(Sbus0.imag, dtype=torch.float64, device="cuda", requires_grad=True) + V1 = solve_power_flow(Sr, Si, max_iter=30, tol=1e-10, grid=grid) + V1.real.sum().backward() + el = pf.sweep._elements + load_bus = el.load_bus[el.load_sel] + expected_lp = -Sr.grad[load_bus] / el.sn_mva + expected_lq = -Si.grad[load_bus] / el.sn_mva + torch.testing.assert_close(lp.grad[0, el.load_sel], expected_lp, atol=1e-7, rtol=1e-5) + torch.testing.assert_close(lq.grad[0, el.load_sel], expected_lq, atol=1e-7, rtol=1e-5) + assert torch.all(lp.grad[1] == 0) # row 1 got no cotangent diff --git a/tests/python/test_scenario_sweep_persistence.py b/tests/python/test_scenario_sweep_persistence.py new file mode 100644 index 0000000..1e391f0 --- /dev/null +++ b/tests/python/test_scenario_sweep_persistence.py @@ -0,0 +1,351 @@ +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at https://mozilla.org/MPL/2.0/. + +"""ScenarioSweepGPU: the batch driver persists across compute() calls. + +compute() picks one of three paths against the live driver (see +ScenarioSweepSession::run in scenario_sweep_session.cu): + + cold : first call, or n_scenarios / batch_size / strategy / cuDSS config / + base state changed -> new driver (allocation + cuDSS ANALYSIS + a + first FACTORIZATION on the first iteration); + warm : only the topology (or the generator mask) changed -> new batch source + on the live driver (CPU connectivity + patch upload), REFACTORIZE only; + hot : only injections / gen_v changed -> one device gather of the new rows. + +The observable proof is in the counters (driver_build_counter / +source_build_counter) and in the one-time timing fields, which a reused driver +reports as 0 (t_analysis_ms, t_alloc_ms, t_first_factorize); results must be +identical to a fresh session's on every path. +""" +import numpy as np +import pytest + +from conftest import requires_gpu +from gpusim2grid._gpusim2grid import have_ls2g_bridge +from test_handle_disconnected_grid import _solved_spur_grid +from test_timings_consistency import _check_to_dict + + +def _check_batch_aggregation(t): + """Like test_timings_consistency's helper, but valid on the bridge path too + (t_ground_truth_check_ms is part of the CPU preprocessing bucket there).""" + assert t.t_cpu_preprocess_ms == pytest.approx( + t.t_preprocess_ms + t.t_ground_truth_check_ms, abs=1e-9) + assert t.t_host_to_device_ms == pytest.approx( + t.t_alloc_ms + t.t_source_init_ms + t.t_branch_data_upload_ms + + t.t_violation_setup_ms, abs=1e-9) + assert t.t_gpu_compute_ms == pytest.approx( + t.t_base_case_solve_only_ms + t.t_analysis_ms + t.t_chunks_total_wall_ms, + abs=1e-9) + assert t.t_grand_total_ms == pytest.approx( + t.t_cpu_preprocess_ms + t.t_host_to_device_ms + t.t_context_init_ms + + t.t_gpu_compute_ms + t.t_device_to_host_ms, abs=1e-9) + +pytestmark = requires_gpu + +needs_bridge = pytest.mark.skipif( + not have_ls2g_bridge, reason="needs the lightsim2grid C++ bridge") + +NB_ITER = 10 +TOL = 1e-10 + + +def _elements(grid): + load_p, load_q = (np.asarray(a, dtype=np.float64) for a in grid.get_loads_res_full()[:2]) + gen_p = np.asarray(grid.get_gen_target_p(), dtype=np.float64) + return load_p, load_q, gen_p + + +def _rows(grid, scales): + load_p, load_q, gen_p = _elements(grid) + scales = np.asarray(scales, dtype=np.float64)[:, None] + n = scales.shape[0] + return (np.tile(load_p, (n, 1)) * scales, np.tile(load_q, (n, 1)) * scales, + np.tile(gen_p, (n, 1))) + + +def _fresh(grid, scales, topology=None, gen_v=None, **kw): + """A brand-new session solving the same rows: the reference for every path.""" + from gpusim2grid import ScenarioSweepGPU + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL, **kw) + sw.set_injections_from_elements(*_rows(grid, scales)) + if topology is not None: + sw.set_topology(topology) + if gen_v is not None: + sw.set_gen_v(gen_v) + V = _np(sw.compute(batch_size=len(scales))) + return V, sw + + +def _np(capsule): + torch = pytest.importorskip("torch") + return torch.from_dlpack(capsule).cpu().numpy().copy() + + +def _one_time_costs_zero(t): + assert t.t_analysis_ms == 0.0 + assert t.t_alloc_ms == 0.0 + assert t.t_context_init_ms == 0.0 + assert t.t_first_factorize.gpu_ms == 0.0 and t.t_first_factorize.wall_ms == 0.0 + + +class TestHotPath: + def test_second_compute_reuses_the_driver(self, ieee14_base_case, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + scales1, scales2 = [1.0, 1.1, 0.9], [0.95, 1.05, 1.2] + + sw.set_injections_from_elements(*_rows(grid, scales1)) + V1 = _np(sw.compute(batch_size=3)) + t1 = sw.timings + assert sw.driver_build_counter == 1 and sw.source_build_counter == 1 + assert t1.t_analysis_ms > 0.0 + assert t1.t_first_factorize.wall_ms > 0.0 + assert t1.n_refactorize == NB_ITER - 1 + _check_batch_aggregation(t1) + _check_to_dict(t1) + + sw.set_injections_from_elements(*_rows(grid, scales2)) + V2 = _np(sw.compute(batch_size=3)) + t2 = sw.timings + assert sw.driver_build_counter == 1 and sw.source_build_counter == 1 + assert sw.run_counter == 2 + _one_time_costs_zero(t2) + assert t2.t_source_init_ms == 0.0 + assert t2.n_refactorize == NB_ITER # refactorize only, no first factorize + assert t2.t_refactorize.wall_ms > 0.0 + _check_batch_aggregation(t2) + _check_to_dict(t2) + + Vref1, _ = _fresh(grid, scales1) + Vref2, _ = _fresh(grid, scales2) + np.testing.assert_allclose(V1, Vref1, atol=solver_atol) + np.testing.assert_allclose(V2, Vref2, atol=solver_atol) + assert not np.allclose(V1, V2) + + def test_results_overwritten_in_place(self, ieee14_base_case): + """v_results_dlpack aliases memory the next (reusing) compute() overwrites.""" + torch = pytest.importorskip("torch") + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(*_rows(grid, [1.0, 1.1])) + view = torch.from_dlpack(sw.compute(batch_size=2)) + snap = view.clone() + sw.set_injections_from_elements(*_rows(grid, [0.9, 1.2])) + sw.compute(batch_size=2) + assert not torch.equal(view, snap) # same memory, new values + assert torch.equal(view, torch.from_dlpack(sw.solver.v_results_dlpack())) + + def test_gen_v_hot_update(self, ieee14_base_case, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + n_gen = len(grid.get_generators()) + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(*_rows(grid, [1.0, 1.05])) + gv1 = np.full((2, n_gen), np.nan); gv1[0, 1] = 1.03 + gv2 = np.full((2, n_gen), np.nan); gv2[1, 2] = 1.02 + sw.set_gen_v(gv1) + V1 = _np(sw.compute(batch_size=2)) + sw.set_gen_v(gv2) + V2 = _np(sw.compute(batch_size=2)) + assert sw.driver_build_counter == 1 and sw.source_build_counter == 1 + _one_time_costs_zero(sw.timings) + np.testing.assert_allclose(V1, _fresh(grid, [1.0, 1.05], gen_v=gv1)[0], atol=solver_atol) + np.testing.assert_allclose(V2, _fresh(grid, [1.0, 1.05], gen_v=gv2)[0], atol=solver_atol) + # Dropping the override: back to the base-case voltages. + sw.solver.clear_gen_v() + V3 = _np(sw.compute(batch_size=2)) + np.testing.assert_allclose(V3, _fresh(grid, [1.0, 1.05])[0], atol=solver_atol) + assert sw.driver_build_counter == 1 + + def test_limit_violations_across_hot_runs(self, ieee14_base_case): + """The fused violation check (whose per-row sentinels are reset every + run) stays correct when the driver is reused.""" + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + n_lines = len(grid.get_lines()) + + def _with_limits(sw): + n_bus, n_bra = sw.n_bus, sw.n_branches + sw.solver.set_limits(np.full(n_bus, np.nan), np.full(n_bus, np.nan), + np.full(n_bra, 0.05), np.full(n_bra, 0.05), n_lines) + sw.compute_limit_violations = True + return sw + + sw = _with_limits(ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL)) + for scales in ([1.0, 1.1], [1.2, 0.8]): + sw.set_injections_from_elements(*_rows(grid, scales)) + sw.compute(batch_size=2) + ref = _with_limits(ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL)) + ref.set_injections_from_elements(*_rows(grid, scales)) + ref.compute(batch_size=2) + for key in ("low_voltage", "high_voltage", "current"): + np.testing.assert_array_equal(sw.get_violation_counts()[key], + ref.get_violation_counts()[key]) + # Same records; the values differ by rounding noise only (a + # refactorized solve vs a first factorization). + for got, exp in zip(sw.get_violations(), ref.get_violations()): + assert len(got) == len(exp) > 0 + for g, e in zip(got, exp): + assert (g.element_type, g.element_id, g.side, g.violation_type, + g.limit) == (e.element_type, e.element_id, e.side, + e.violation_type, e.limit) + assert g.value == pytest.approx(e.value, rel=1e-9) + assert sw.driver_build_counter == 1 + + +class TestWarmPath: + @needs_bridge + def test_new_topology_rebuilds_only_the_source(self, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid, _, spur_line, _ = _solved_spur_grid(distributed_slack=False) + scales = [1.0, 1.05, 0.95, 1.1] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(*_rows(grid, scales)) + + topo_a = [[], [3], [], [int(spur_line)]] # row 3 islands the spur bus + sw.set_topology(topo_a) + Va = _np(sw.compute(batch_size=4)) + assert list(sw.get_disconnected()) == [0, 0, 0, 1] + assert np.all(np.isnan(Va[3])) + assert sw.driver_build_counter == 1 and sw.source_build_counter == 1 + + topo_b = [[int(spur_line)], [], [5], []] # row 0 islands, row 3 recovers + sw.set_topology(topo_b) + Vb = _np(sw.compute(batch_size=4)) + assert list(sw.get_disconnected()) == [1, 0, 0, 0] + assert np.all(np.isnan(Vb[0])) and np.all(np.isfinite(Vb[3])) + assert sw.driver_build_counter == 1 and sw.source_build_counter == 2 + t = sw.timings + _one_time_costs_zero(t) + assert t.t_preprocess_ms > 0.0 or t.t_source_init_ms >= 0.0 # source rebuilt + assert t.n_refactorize == NB_ITER + _check_batch_aggregation(t) + + Vref_a = _fresh(grid, scales, topology=topo_a)[0] + Vref_b = _fresh(grid, scales, topology=topo_b)[0] + np.testing.assert_allclose(Va[:3], Vref_a[:3], atol=solver_atol) + np.testing.assert_allclose(Vb[1:], Vref_b[1:], atol=solver_atol) + + @needs_bridge + def test_handle_disconnected_grid_warm_path(self, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid, _, spur_line, spur_bus = _solved_spur_grid(distributed_slack=False) + me_to_solver = np.asarray(grid.id_me_to_ac_solver()) + spur_solver = int(me_to_solver[spur_bus]) if me_to_solver.size else int(spur_bus) + scales = [1.0, 1.1] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL, + handle_disconnected_grid=True) + sw.set_injections_from_elements(*_rows(grid, scales)) + sw.set_topology([[], []]) + V0 = _np(sw.compute(batch_size=2)) + sw.set_topology([[int(spur_line)], []]) + V1 = _np(sw.compute(batch_size=2)) + assert sw.driver_build_counter == 1 and sw.source_build_counter == 2 + assert list(sw.get_disconnected()) == [0, 0] + assert np.isnan(V1[0, spur_solver]) and np.all(np.isfinite(V1[1])) + ref0 = _fresh(grid, scales, topology=[[], []], handle_disconnected_grid=True)[0] + ref1 = _fresh(grid, scales, topology=[[int(spur_line)], []], + handle_disconnected_grid=True)[0] + np.testing.assert_allclose(V0, ref0, atol=solver_atol) + np.testing.assert_allclose(V1[1], ref1[1], atol=solver_atol) + finite = np.isfinite(ref1[0]) + np.testing.assert_allclose(V1[0][finite], ref1[0][finite], atol=solver_atol) + + +class TestColdPath: + def test_shape_and_config_changes_rebuild(self, ieee14_base_case, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + + sw.set_injections_from_elements(*_rows(grid, [1.0, 1.1])) + sw.compute(batch_size=2) + assert sw.driver_build_counter == 1 + + # n_scenarios changed -> cold + sw.set_injections_from_elements(*_rows(grid, [1.0, 1.1, 0.9])) + V = _np(sw.compute(batch_size=3)) + assert sw.driver_build_counter == 2 + assert sw.timings.t_analysis_ms > 0.0 + np.testing.assert_allclose(V, _fresh(grid, [1.0, 1.1, 0.9])[0], atol=solver_atol) + + # strategy changed -> cold + sw.strategy = "direct_iter0_only" + sw.compute(batch_size=3) + assert sw.driver_build_counter == 3 + + # reordering changed -> cold + sw.reordering_alg = "amd" + sw.compute(batch_size=3) + assert sw.driver_build_counter == 4 + + # nb_iter changed -> NOT a rebuild, but takes effect + sw.nb_iter = NB_ITER + 3 + sw.compute(batch_size=3) + assert sw.driver_build_counter == 4 + assert sw.timings.nb_iter == NB_ITER + 3 + + # batch_size changed -> cold + sw.compute(batch_size=2) + assert sw.driver_build_counter == 5 + + @needs_bridge + def test_fixed_batch_capacity_keeps_one_chunk(self, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid, _, spur_line, _ = _solved_spur_grid(distributed_slack=False) + scales = [1.0, 1.05, 0.95, 1.1, 1.02] + topo = [[int(spur_line)], [], [int(spur_line)], [], []] # 2 islanded rows + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.solver.fixed_batch_capacity = True + sw.set_injections_from_elements(*_rows(grid, scales)) + sw.set_topology(topo) + V = _np(sw.compute(batch_size=5)) + t = sw.timings + assert t.n_chunks == 1 + assert t.chunk_size == 5 and sw.solver.used_batch_size == 5 + assert sw.solver.capacity == 5 and sw.solver.n_active == 3 + np.testing.assert_array_equal(sw.active_to_orig, [1, 3, 4]) + ref = _fresh(grid, scales, topology=topo)[0] + for r in (1, 3, 4): + np.testing.assert_allclose(V[r], ref[r], atol=solver_atol) + assert np.all(np.isnan(V[0])) and np.all(np.isnan(V[2])) + + +class TestDevicePath: + def test_set_injections_dlpack_matches_numpy(self, ieee14_base_case, solver_atol): + torch = pytest.importorskip("torch") + from gpusim2grid import ScenarioSweepGPU + from gpusim2grid._gpusim2grid import is_fp32 + from gpusim2grid._ls2g_utils import build_bus_injections + grid = ieee14_base_case["grid"] + scales = [1.0, 1.1, 0.9] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + el = sw._elements + p_mw, q_mvar = build_bus_injections(el, *_rows(grid, scales)) + S = torch.tensor((p_mw + 1j * q_mvar) / el.sn_mva, + dtype=torch.complex64 if is_fp32 else torch.complex128, + device="cuda") + sw.solver.set_injections_dlpack(S.__dlpack__(), torch.cuda.current_stream().cuda_stream) + V_dev = _np(sw.compute(batch_size=3)) + ref = _fresh(grid, scales)[0] + np.testing.assert_allclose(V_dev, ref, atol=solver_atol) + + # The tensor was consumed by copy: freeing it changes nothing. + del S + torch.cuda.synchronize() + V_again = _np(sw.compute(batch_size=3)) + np.testing.assert_allclose(V_again, ref, atol=solver_atol) + assert sw.driver_build_counter == 1 + + # Validation. + bad = torch.zeros(3, sw.n_bus + 1, dtype=torch.complex128, device="cuda") + with pytest.raises(RuntimeError, match="extent"): + sw.solver.set_injections_dlpack(bad.__dlpack__()) + wrong_dtype = torch.zeros(3, sw.n_bus, dtype=torch.float64, device="cuda") + with pytest.raises(RuntimeError, match="dtype"): + sw.solver.set_injections_dlpack(wrong_dtype.__dlpack__()) From 73e60fe20bb60eb1b75b5243178d367e7af70d67 Mon Sep 17 00:00:00 2001 From: DONNOT Benjamin Date: Fri, 11 Sep 2026 17:52:24 +0200 Subject: [PATCH 2/7] add generator disconnection Signed-off-by: DONNOT Benjamin --- CLAUDE.md | 2 +- examples/batch_differentiable_pf.py | 86 +++++ src/_cpp/contingency/batch_pf_driver.cu | 36 +- src/_cpp/python_bindings.cpp | 5 + src/_cpp/scenario_sweep_session.cu | 1 + src/_cpp/scenario_sweep_session.hpp | 6 + src/gpusim2grid/differentiable/_batch_pf.py | 155 +++++++- src/gpusim2grid/scenario_sweep/__init__.py | 7 + tests/python/test_batch_power_flow.py | 374 ++++++++++++++++++-- 9 files changed, 633 insertions(+), 39 deletions(-) create mode 100644 examples/batch_differentiable_pf.py diff --git a/CLAUDE.md b/CLAUDE.md index 2be5b18..687005a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -154,7 +154,7 @@ Three construction-time-only knobs select cuDSS's analysis-phase behavior, expos `src/gpusim2grid/differentiable/` wraps `AcPfNrSession` in a `torch.autograd.Function` (`PowerFlowFunction`, `solve_power_flow`). Backward uses the adjoint method (implicit function theorem): it solves Jᵀλ = x̄ via `AcPfNrSession.solve_JT_dlpack` reusing the converged factorization. Sbus is split into real/imag tensors because autograd needs real tensors. Sign conventions and the x-space projection are documented at the top of `_power_flow_op.py` — read it before touching gradients. -**`BatchPowerFlow`** (`differentiable/_batch_pf.py`) is the batched, ML-facing entry point: built from a solved lightsim2grid grid, it owns a `ScenarioSweepGPU` and takes lightsim2grid-`ScenarioSweep`-style per-element inputs — `load_p`, `load_q`, `gen_p`, `gen_v` (all differentiable) and the boolean `line_status`/`trafo_status` masks (**True = connected**, grid2op's convention, unlike lightsim2grid's `set_contingency_lines` where True = tripped). The element→bus assembly is plain torch (`index_add` mirroring `build_bus_injections`), so autograd gives the `load_p`/`load_q`/`gen_p` VJPs for free; an internal `_BatchPowerFlowOp` maps `(P, Q, gen_v) → V` and its backward does the batched adjoint. Its gradient is verified against finite differences in `tests/python/test_batch_power_flow.py` (gradcheck + explicit central differences for every input — keep it that way when touching any of this). The `gen_v` gradient has a direct term (`Re(e^{-jθ}·ḡV)` at the Vm-fixed bus, Python side) and an indirect one (`−λ·∂S_calc/∂Vm_k`, the dS/dVm column `fill_J` never stores for a Vm-fixed bus, computed per slot on the patched Ybus by `gen_v_adjoint_kernel` through a Ybus transpose-position map); only generators whose own bus is Vm-fixed get a non-zero gradient, NaN entries get 0. `torch.autograd.gradcheck` runs many forwards before it back-propagates the first output, so it needs `snapshot_jacobian=True` (the default alias mode refuses that pattern through the `run_counter` guard). +**`BatchPowerFlow`** (`differentiable/_batch_pf.py`) is the batched, ML-facing entry point: built from a solved lightsim2grid grid, it owns a `ScenarioSweepGPU` and takes lightsim2grid-`ScenarioSweep`-style per-element inputs — `load_p`, `load_q`, `gen_p`, `gen_v` (all differentiable) and the boolean `line_status`/`trafo_status` masks (**True = connected**, grid2op's convention, unlike lightsim2grid's `set_contingency_lines` where True = tripped). The element→bus assembly is plain torch (`index_add` mirroring `build_bus_injections`), so autograd gives the `load_p`/`load_q`/`gen_p` VJPs for free; an internal `_BatchPowerFlowOp` maps `(P, Q, gen_v) → V` and its backward does the batched adjoint. Its gradient is verified against finite differences in `tests/python/test_batch_power_flow.py` (gradcheck + explicit central differences for every input — keep it that way when touching any of this). The `gen_v` gradient has a direct term (`Re(e^{-jθ}·ḡV)` at the Vm-fixed bus, Python side) and an indirect one (`−λ·∂S_calc/∂Vm_k`, the dS/dVm column `fill_J` never stores for a Vm-fixed bus, computed per slot on the patched Ybus by `gen_v_adjoint_kernel` through a Ybus transpose-position map); only generators whose own bus is Vm-fixed get a non-zero gradient, NaN entries get 0. `torch.autograd.gradcheck` runs many forwards before it back-propagates the first output, so it needs `snapshot_jacobian=True` (the default alias mode refuses that pattern through the `run_counter` guard). A snapshot also requires the session's batch *source* to be the one of its own forward (active-slot map and mask streams are read from the live source at backward time): a hot forward in between is fine, a topology/`gen_status`/row-count change is refused through the `source_build_counter` guard. **`gen_status`** (`(n_scen, n_gen)` bool, True = connected) is the third, discrete contingency input: it maps to `ScenarioSweepGPU.set_contingency_gens`, with the injection correction done in torch (the off generator's P is not added, a non-regulating one's target Q leaves the constant term through `InjectionElements.gen_target_q_mvar`/`gen_vreg_on`, and its `gen_v` column is NaN'd so it can never clobber a still-connected co-located generator). Gradient rules that follow: an off generator's `gen_p`/`gen_v` get 0; on a row where a bus is *released* (PV→PQ) its `load_q` gradient is live and any `gen_v` there is 0 (`ScenarioSweepSession::get_row_pv_to_pq`, captured per forward); on a row where a reserved bus is still PV, `BatchPfDriver::solve_JT_batch` zeroes λ at every identity row of the chunk's mask stream (`zero_identity_rows_kernel`, masked P/Q rows and pinned Q rows alike) right after the cuDSS solve — an identity row `e_c` makes λ_r absorb `x̄_c − Σ J[r',c] λ_r'` while leaving every other λ exactly the reduced system's, and that λ_r would otherwise surface as a Q gradient on a frozen equation and be contracted by `gen_v_adjoint_kernel` with `∂Q_k/∂Vm_j` for the neighbours. **Call-to-call state**: a `forward()` without `line_status`/`trafo_status` (resp. `gen_status`) after one that had them hands the session an all-empty trip list (warm source rebuild) resp. an all-False mask (cold, since the reserved structure is released) — the same inputs always give the same answer whatever ran before (`TestCallToCallState`). The C++ pieces this rests on (all `ScenarioSweepSession`, DLPack in/out so a JAX front-end can reuse them): diff --git a/examples/batch_differentiable_pf.py b/examples/batch_differentiable_pf.py new file mode 100644 index 0000000..8ab2de1 --- /dev/null +++ b/examples/batch_differentiable_pf.py @@ -0,0 +1,86 @@ +# This Source Code Form is subject to the terms of the Mozilla Public +# License, v. 2.0. If a copy of the MPL was not distributed with this +# file, You can obtain one at https://mozilla.org/MPL/2.0/. + +""" +batch_differentiable_pf.py — a batch of power flows as a differentiable +PyTorch layer, driven by lightsim2grid-style per-element inputs. + +``BatchPowerFlow`` takes, per scenario (row), the load / generator injections, +the generator voltage set-points and the line / trafo statuses, solves the whole +batch in ONE GPU pass, and back-propagates through it with the adjoint method. +Here a tiny model learns a generator redispatch + voltage set-points that keep +every bus of the IEEE 14-bus grid near 1.0 pu under random load scalings and +random N-1 line trips. + +The point of the timings printed at the end: the first call builds the GPU +batch driver (cuDSS analysis + first factorization); every later call with the +same batch size reuses it (only the new rows move to the GPU, the Jacobians are +refactorized), and the transposed system the backward needs is built once, on +the first backward, then only refactorized. + +Requires PyTorch with CUDA. Run: + python examples/batch_differentiable_pf.py +""" +import numpy as np + +from _common import load_case +from gpusim2grid.compilation_options import is_fp32 + + +def main(): + import torch + + if not torch.cuda.is_available(): + raise SystemExit("This example requires a CUDA-capable PyTorch build.") + if is_fp32: + print("NOTE: gpusim2grid was built in FP32; gradients are most accurate " + "against an FP64 build.") + + from gpusim2grid.differentiable import BatchPowerFlow + + grid = load_case("case14")["grid"] + pf = BatchPowerFlow.from_lsgrid(grid, nb_iter=8, tol_base=1e-10) + dev = pf.device + rdt = torch.float32 if is_fp32 else torch.float64 + + n_scen = 64 + rng = np.random.default_rng(0) + # Random load scalings and one random line trip per row (True = connected). + scales = torch.tensor(rng.uniform(0.8, 1.2, size=(n_scen, 1)), dtype=rdt, device=dev) + load_p = pf._load_p_base[None, :] * scales + load_q = pf._load_q_base[None, :] * scales + line_status = torch.ones(n_scen, pf.n_line, dtype=torch.bool, device=dev) + line_status[torch.arange(n_scen), torch.tensor(rng.integers(2, 12, size=n_scen))] = False + + # Learnable per-generator redispatch (MW) and voltage set-points (pu), + # shared across the batch; the slack absorbs the mismatch. + dp = torch.zeros(pf.n_gen, dtype=rdt, device=dev, requires_grad=True) + vset = torch.full((pf.n_gen,), 1.04, dtype=rdt, device=dev, requires_grad=True) + opt = torch.optim.Adam([dp, vset], lr=5e-3) + + for step in range(15): + opt.zero_grad() + gen_p = (pf._gen_p_base + dp)[None, :].expand(n_scen, -1) + gen_v = vset[None, :].expand(n_scen, -1) + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=gen_v, + line_status=line_status) + valid = torch.isfinite(V.real) # islanded rows are NaN + vm = torch.where(valid, V, torch.ones_like(V)).abs() + loss = ((vm - 1.0) ** 2 * valid).sum() / valid.sum() + 1e-4 * (dp ** 2).sum() + loss.backward() + opt.step() + t = pf.timings + print(f"step {step:2d} loss {loss.item():.3e} " + f"driver builds {pf.sweep.driver_build_counter} " + f"analysis {t.t_analysis_ms:6.2f} ms first-factorize {t.t_first_factorize.wall_ms:5.2f} ms " + f"refactorize x{t.n_refactorize} " + f"adjoint: analysis x{t.adjoint_n_analysis} factorize x{t.adjoint_n_factorize} " + f"refactorize x{t.adjoint_n_refactorize} solve x{t.adjoint_n_solve}") + + print(f"learned redispatch (MW): {dp.detach().cpu().numpy().round(3)}") + print(f"learned set-points (pu): {vset.detach().cpu().numpy().round(4)}") + + +if __name__ == "__main__": + main() diff --git a/src/_cpp/contingency/batch_pf_driver.cu b/src/_cpp/contingency/batch_pf_driver.cu index 15fff67..23f6e13 100644 --- a/src/_cpp/contingency/batch_pf_driver.cu +++ b/src/_cpp/contingency/batch_pf_driver.cu @@ -948,6 +948,27 @@ __global__ void transpose_gather_values_kernel( dst[b * nnz + perm[i]] = src[b * nnz + i]; } +// Zero the adjoint vector at this chunk's identity rows (handle_disconnected_ +// grid masked P/Q rows and the PV-pinned Q rows of generator contingencies). +// An identity row r = e_c decouples: λ_r appears in the single Jᵀ equation of +// its column c, so it absorbs x̄_c − Σ_{r'≠r} J[r',c] λ_{r'} and every other λ +// is exactly the reduced (unpinned) system's. That λ_r is not a real +// sensitivity -- the equation is frozen, F_r ≡ 0 whatever Sbus -- and it must +// not reach the outputs: as λ_{q_k} it would be reported as a Q gradient at a +// bus whose Q equation is off, and gen_v_adjoint_kernel would contract it +// with ∂Q_k/∂Vm_j for every neighbour j (a row the bare system never has). +// For a masked row the value is already 0 (its column carries no live entry). +__global__ void zero_identity_rows_kernel( + cuda_real_type* __restrict__ d_sol, // [batch × dim_J] (slot order) + const int* __restrict__ d_mask_slot, + const int* __restrict__ d_mask_row, + int dim_J, int n_entries) +{ + const int i = blockIdx.x * blockDim.x + threadIdx.x; + if (i >= n_entries) return; + d_sol[static_cast(d_mask_slot[i]) * dim_J + d_mask_row[i]] = cuda_real_type(0); +} + // gen_v adjoint: for every (slot, Vm-fixed bus k) the contraction of the // adjoint vector with the dS_calc/dVm_k column (which J never stores for a // Vm-fixed bus): @@ -1179,11 +1200,24 @@ void BatchPfDriver::solve_JT_batch( dim_J, n_active_, /*zero_nonfinite=*/true, cs); CHK_CUDA_BPF(cudaGetLastError()); - // ③ Solve and scatter back to original order. + // ③ Solve, drop the identity-row components (see zero_identity_rows_kernel; + // the driver is one chunk here -- keep_final_jacobian requires it -- so + // the chunk-0 slice is the whole batch) and scatter back to original order. timer.start(); A.solver.solve(); A.t_solve += timer.stop_ms(); ++A.n_solve; + { + NrIterBuffers mb{}; + source_.fill_mask_buffers(mb, /*chunk_idx=*/0, + thrust::raw_pointer_cast(base.d_J_outer.data())); + if (mb.n_mask_rows > 0) { + zero_identity_rows_kernel<<<(mb.n_mask_rows + BS - 1) / BS, BS, 0, cs>>>( + thrust::raw_pointer_cast(A.d_sol.data()), + mb.d_mask_slot, mb.d_mask_row, dim_J, mb.n_mask_rows); + CHK_CUDA_BPF(cudaGetLastError()); + } + } zero_d(A.d_sol_full, cs); launch_scatter_rows(thrust::raw_pointer_cast(A.d_sol_full.data()), thrust::raw_pointer_cast(A.d_sol.data()), d_map, diff --git a/src/_cpp/python_bindings.cpp b/src/_cpp/python_bindings.cpp index 118944e..4a095da 100644 --- a/src/_cpp/python_bindings.cpp +++ b/src/_cpp/python_bindings.cpp @@ -1435,6 +1435,11 @@ PYBIND11_MODULE(_gpusim2grid, m) "some from set_contingency_gens' mask).") .def_property_readonly("has_gen_contingency", &ScenarioSweepSession::has_gen_contingency, "True once set_contingency_gens() has been called.") + .def("get_row_pv_to_pq", &ScenarioSweepSession::get_row_pv_to_pq, + "list[list[int]]: per scenario (original row order), the AC-solver " + "buses the last run() turned PV->PQ because set_contingency_gens' " + "mask took out every generator locally regulating them. All empty " + "without a mask; empty before run().") .def("run", &ScenarioSweepSession::run, "Solve all scenarios. Fills the device-side voltage and residual " "buffers. Requires set_injections() first. A scenario whose " diff --git a/src/_cpp/scenario_sweep_session.cu b/src/_cpp/scenario_sweep_session.cu index 10fc3c3..a1339c4 100644 --- a/src/_cpp/scenario_sweep_session.cu +++ b/src/_cpp/scenario_sweep_session.cu @@ -662,6 +662,7 @@ void ScenarioSweepSession::run() _prepare_gen_contingency(required, row_pv_to_pq, row_slack_off); if (required != reserved_buses_) _build_base_state(required); + row_pv_to_pq_ = row_pv_to_pq; if (has_gen_off_ && strategy_type_ == ContingencySolverType::DirectBaseCaseFactors) { diff --git a/src/_cpp/scenario_sweep_session.hpp b/src/_cpp/scenario_sweep_session.hpp index f099a5a..d106634 100644 --- a/src/_cpp/scenario_sweep_session.hpp +++ b/src/_cpp/scenario_sweep_session.hpp @@ -262,6 +262,11 @@ struct ScenarioSweepSession { BoolMat gen_off_; bool has_gen_off_ = false; std::vector reserved_buses_; + // Per-row buses turned PV->PQ by the last run() (ORIGINAL row order, + // n_scenarios entries; all empty without a generator mask). Exposed for the + // differentiable wrapper: a released bus' voltage set-point is only an NR + // start value there, so its gen_v gradient is 0 on that row. + std::vector> row_pv_to_pq_; Eigen::SparseMatrix Ybus_cm_; CplxVect Vinit_, Sbus_; @@ -415,6 +420,7 @@ struct ScenarioSweepSession { // caller observe when run() rebuilt the base state. int dim_J() const; std::vector get_reserved_buses() const { return reserved_buses_; } + std::vector> get_row_pv_to_pq() const { return row_pv_to_pq_; } bool has_gen_contingency() const { return has_gen_off_; } // ========================================================================= diff --git a/src/gpusim2grid/differentiable/_batch_pf.py b/src/gpusim2grid/differentiable/_batch_pf.py index 550ce03..019b709 100644 --- a/src/gpusim2grid/differentiable/_batch_pf.py +++ b/src/gpusim2grid/differentiable/_batch_pf.py @@ -8,22 +8,51 @@ pf = BatchPowerFlow.from_lsgrid(grid, nb_iter=6) V = pf(load_p=..., load_q=..., gen_p=..., gen_v=..., - line_status=..., trafo_status=...) # complex (n_scen, n_bus) + line_status=..., trafo_status=..., gen_status=...) # complex (n_scen, n_bus) Every input is a ``(n_scenarios, n_elements)`` tensor (or ``None`` = the grid's own base-case value in every row); row ``i`` is one independent scenario: its own injections, its own voltage set-points and its own set of -disconnected branches (``line_status`` / ``trafo_status`` are boolean masks -with **True = connected**, grid2op's convention; a row trips the branches -whose mask is False). The whole batch is solved in ONE GPU pass by a -:class:`gpusim2grid.ScenarioSweepGPU` session that this object keeps alive, -so consecutive calls with the same number of rows reuse everything: no -cuDSS analysis, no first factorization, no re-upload of the grid, only the -new rows move to the device (see ``ScenarioSweepGPU.driver_build_counter``). +disconnected branches and generators (``line_status`` / ``trafo_status`` / +``gen_status`` are boolean masks with **True = connected**, grid2op's +convention; a row trips the elements whose mask is False). The whole batch +is solved in ONE GPU pass by a :class:`gpusim2grid.ScenarioSweepGPU` session +that this object keeps alive, so consecutive calls with the same number of +rows reuse everything: no cuDSS analysis, no first factorization, no +re-upload of the grid, only the new rows move to the device (see +``ScenarioSweepGPU.driver_build_counter``). Differentiable inputs: ``load_p``, ``load_q``, ``gen_p`` (through Sbus) and ``gen_v`` (through the seeded voltage magnitude). ``line_status`` / -``trafo_status`` are discrete and carry no gradient. +``trafo_status`` / ``gen_status`` are discrete and carry no gradient. + +Generator contingencies (``gen_status``) are lightsim2grid's +``ScenarioSweep.set_contingency_gens`` (see ``ScenarioSweepGPU``): a +disconnected generator's P leaves Sbus (and the target Q of a non +voltage-regulating one), the distributed slack is re-weighted without it, +and when the LAST generator locally regulating a bus is off that bus turns +PV->PQ for that row only. Structurally, every bus that can flip in some row +gets a reserved Vm column + Q equation (the session rebuilds its base state +whenever that set changes: one cuDSS analysis, a cold driver); rows where +such a bus is still PV identity-pin its Q row. For the gradient this means: +the ``gen_p`` / ``gen_v`` of a disconnected generator get 0 (its ``gen_v`` +is passed to the session as NaN, "keep the base voltage", so it can never +clobber a still-connected co-located generator's set-point either); on a +row where a bus is released, ``load_q`` at that bus has a gradient and the +``gen_v`` of any generator there has none (the set-point is only an NR start +value); on a row where a reserved bus is still PV, the batched adjoint drops +the frozen Q equation's component of λ (``zero_identity_rows_kernel``) so it +is neither reported as a Q gradient nor contracted into the neighbours' +``gen_v`` gradients -- with that, λ restricted to the live equations is +exactly the bare system's. + +Call-to-call state: a call without ``line_status``/``trafo_status`` (or +without ``gen_status``) after one that had them explicitly resets the +session's topology (an all-empty trip list per row; a warm source rebuild) +or generator mask (an all-False mask, which also releases the reserved +structure: a cold rebuild). So the same inputs always give the same answer +whatever the previous call was -- verified in +``tests/python/test_batch_power_flow.py``. Adjoint math (batched) ---------------------- @@ -75,7 +104,11 @@ forward→forward→backward(first) pattern into a clear error; pass ``snapshot_jacobian=True`` to clone those buffers in every forward instead (one D2D copy of ``capacity x nnz_J`` reals and ``capacity x nnz_Y`` -complex, kept until backward). +complex, kept until backward). A snapshot still needs the session's *batch +source* of its own forward (the active-slot map, the identity-row masks, the +pinned rows all live there and are read at backward time): another forward +in between may change the injections / ``gen_v`` (hot path) but not the +topology or generator mask (``source_build_counter`` guard). """ import numpy as np @@ -135,6 +168,15 @@ def __init__(self, sweep, *, snapshot_jacobian=False): self._gen_bus_all = torch.as_tensor(np.asarray(el.gen_bus, dtype=np.int64), device=dev) self._gen_bus_np = np.ascontiguousarray(el.gen_bus, dtype=np.int32) self._is_vm_fixed = torch.as_tensor(self._solver.is_vm_fixed_bus, device=dev) + # Generator contingencies (build_bus_injections' gen_off correction): + # the target Q of a NON-regulating generator sits inside const_mw and + # leaves with it. Older snapshots (no reactive data) cannot do this. + if el.gen_target_q_mvar is not None and el.gen_vreg_on is not None: + q_off = np.where(np.asarray(el.gen_vreg_on, dtype=bool), 0.0, + np.asarray(el.gen_target_q_mvar, dtype=np.float64)) + self._gen_q_off = torch.as_tensor(q_off, dtype=rdt, device=dev) + else: + self._gen_q_off = None # Base-case per-element values: what a ``None`` input means. self._load_p_base = torch.tensor(np.array(el.load_p_base), dtype=rdt, device=dev) @@ -157,6 +199,10 @@ def __init__(self, sweep, *, snapshot_jacobian=False): self._topology_in_session = False self._pending_topology = None # ragged list to hand to the session on the next run self._gen_v_in_session = False + self._gen_off_mask = None # (n_scen, n_gen) bool, True = disconnected + self._gen_off_in_session = False + self._pending_gen_off = None # mask to hand to the session on the next run + self._released = None # (n_scen, n_bus) bool of the last run, or None # ------------------------------------------------------------------ build @classmethod @@ -193,7 +239,7 @@ def __call__(self, *args, **kwargs): return self.forward(*args, **kwargs) def forward(self, load_p=None, load_q=None, gen_p=None, gen_v=None, - line_status=None, trafo_status=None): + line_status=None, trafo_status=None, gen_status=None): """Solve one scenario per row; returns complex ``V`` ``(n_scen, n_bus)`` on the GPU (per-unit, AC-solver bus numbering; NaN rows = scenarios the connectivity pre-check dropped). @@ -204,16 +250,22 @@ def forward(self, load_p=None, load_q=None, gen_p=None, gen_v=None, = keep the base-case voltage for that (row, gen)) line_status : (n_scen, n_line) bool, True = connected trafo_status : (n_scen, n_trafo) bool, True = connected + gen_status : (n_scen, n_gen) bool, True = connected (generator + contingencies, see the module docstring; needs the + lightsim2grid bridge, and only generators regulating + their own bus may be disconnected) ``None`` = the grid's base-case value in every row. At least one input must be given (it fixes n_scen). """ - n_scen = self._infer_n_scen(load_p, load_q, gen_p, gen_v, line_status, trafo_status) + n_scen = self._infer_n_scen(load_p, load_q, gen_p, gen_v, + line_status, trafo_status, gen_status) load_p = self._as_input(load_p, self.n_load, self._load_p_base, n_scen, "load_p") load_q = self._as_input(load_q, self.n_load, self._load_q_base, n_scen, "load_q") gen_p = self._as_input(gen_p, self.n_gen, self._gen_p_base, n_scen, "gen_p") gen_v = None if gen_v is None else self._as_input(gen_v, self.n_gen, None, n_scen, "gen_v") self._apply_topology(line_status, trafo_status, n_scen) + gen_off = self._apply_gen_status(gen_status, n_scen) if n_scen != self._last_n_scen: # Capacity == n_scen: one chunk, whatever rows get islanded. @@ -224,6 +276,18 @@ def forward(self, load_p=None, load_q=None, gen_p=None, gen_v=None, inv_sn = 1.0 / self.sn_mva P = self._const_re.unsqueeze(0).expand(n_scen, -1).clone() Q = self._const_im.unsqueeze(0).expand(n_scen, -1).clone() + if gen_off is not None: + # Same operands as build_bus_injections(gen_off=...): the P a + # disconnected generator would have added is not added, the target + # Q of a non-regulating one is taken back out of the constant term, + # and its gen_v set-point is dropped (NaN = keep the base voltage). + gen_p = torch.where(gen_off, torch.zeros_like(gen_p), gen_p) + if self._gen_sel.numel(): + q_off = torch.where(gen_off, self._gen_q_off.unsqueeze(0).expand(n_scen, -1), + torch.zeros_like(gen_p)) + Q = Q.index_add(1, self._gen_bus_sel, -q_off[:, self._gen_sel] * inv_sn) + if gen_v is not None: + gen_v = torch.where(gen_off, torch.full_like(gen_v, float("nan")), gen_v) if self._gen_sel.numel(): P = P.index_add(1, self._gen_bus_sel, gen_p[:, self._gen_sel] * inv_sn) if self._load_sel.numel(): @@ -237,6 +301,13 @@ def get_disconnected(self): """(n_scen,) int: 1 where the last call dropped the row (NaN).""" return self._sweep.get_disconnected() + @property + def reserved_switchable_buses(self): + """(k,) int ndarray: AC-solver buses currently owning a reserved Vm + column + Q equation for ``gen_status`` (empty without generator + contingencies) -- see ``ScenarioSweepGPU.reserved_switchable_buses``.""" + return self._sweep.reserved_switchable_buses + def last_residuals(self): """(n_scen,) float: ``‖F‖∞`` of each row after the last call.""" return self._solver.residuals.to_numpy() @@ -286,7 +357,7 @@ def _infer_n_scen(self, *inputs): if n is None: raise ValueError( "BatchPowerFlow needs at least one input (load_p, load_q, gen_p, gen_v, " - "line_status or trafo_status) to know the number of scenarios") + "line_status, trafo_status or gen_status) to know the number of scenarios") if n <= 0: raise ValueError("the number of scenarios must be > 0") return n @@ -349,6 +420,33 @@ def _apply_topology(self, line_status, trafo_status, n_scen): self._pending_topology = ragged self._topology_mask = tripped.clone() + def _apply_gen_status(self, gen_status, n_scen): + """Same contract as _apply_topology for the generator mask. Returns the + (n_scen, n_gen) bool "disconnected" mask to apply to the injections + (None when no generator is off), and queues the session mask when it + must change: a call without gen_status after one with it hands an + all-False mask over (bit-identical to no mask for the session, which + also drops the reserved structure it no longer needs).""" + self._pending_gen_off = None + if gen_status is None: + if self._gen_off_mask is not None or ( + self._gen_off_in_session and n_scen != self._last_n_scen): + self._pending_gen_off = np.zeros((n_scen, self.n_gen), dtype=bool) + self._gen_off_mask = None + return None + + gen_off = ~self._as_status(gen_status, self.n_gen, n_scen, "gen_status") + if self._gen_q_off is None: + raise RuntimeError( + "gen_status needs the generators' reactive set-points; rebuild the " + "ScenarioSweepGPU from a lightsim2grid grid with the current " + "extract_injection_elements().") + prev = self._gen_off_mask + if prev is None or prev.shape != gen_off.shape or not torch.equal(prev, gen_off): + self._pending_gen_off = np.ascontiguousarray(gen_off.cpu().numpy(), dtype=bool) + self._gen_off_mask = gen_off.clone() + return gen_off if bool(gen_off.any()) else None + def _device_index(sweep): """CUDA ordinal the session lives on (the facade normalises ``device``).""" @@ -375,6 +473,14 @@ def forward(ctx, P: Tensor, Q: Tensor, gen_v, pf: BatchPowerFlow) -> Tensor: pf._sweep.set_topology(pf._pending_topology) pf._topology_in_session = True pf._pending_topology = None + if pf._pending_gen_off is not None: + # After the injections: the session checks the row counts against + # them. The facade validates the mask (and would re-assemble Sbus + # only for set_injections_from_elements inputs, which this path + # never uses -- the injections above already carry the correction). + pf._sweep.set_contingency_gens(pf._pending_gen_off) + pf._gen_off_in_session = True + pf._pending_gen_off = None if gen_v is not None: solver.set_gen_v_dlpack(gen_v.detach().contiguous().__dlpack__(), pf._gen_bus_np, stream) pf._gen_v_in_session = True @@ -389,8 +495,22 @@ def forward(ctx, P: Tensor, Q: Tensor, gen_v, pf: BatchPowerFlow) -> Tensor: V = torch.from_dlpack(solver.v_results_dlpack()).clone() + # Rows where a generator contingency released a bus (PV->PQ): its + # gen_v is only an NR start value there -> zero gradient. + pf._released = None + if pf._gen_off_in_session and want_gen_v: + rel = solver.get_row_pv_to_pq() + if any(rel): + m = torch.zeros(V.shape, dtype=torch.bool, device=dev) + for r, buses in enumerate(rel): + if buses: + m[r, torch.as_tensor(buses, device=dev)] = True + pf._released = m + ctx.pf = pf ctx.run_id = solver.run_counter + ctx.source_id = solver.source_build_counter + ctx.released = pf._released ctx.want_gen_v = bool(want_gen_v) ctx.J = ctx.Y = None if needs_grad and pf.snapshot_jacobian: @@ -415,6 +535,12 @@ def backward(ctx, grad_V: Tensor): "gradient belongs to, overwriting its Jacobians on the GPU. Call " "backward() before the next forward(), or build the model with " "snapshot_jacobian=True to keep a copy per forward.") + if ctx.J is not None and solver.source_build_counter != ctx.source_id: + raise RuntimeError( + "BatchPowerFlow.backward: another forward() with a different " + "topology, generator mask or number of rows ran after the one this " + "gradient belongs to; the Jacobian snapshot no longer matches the " + "session's batch structure. Call backward() before such a forward().") theta_col = torch.as_tensor(solver.theta_col_of_bus, device=dev) vm_col = torch.as_tensor(solver.vm_col_of_bus, device=dev) @@ -466,6 +592,9 @@ def backward(ctx, grad_V: Tensor): ok = (gen_bus >= 0) & pf._is_vm_fixed[safe_bus] g = g_bus[:, safe_bus] g = torch.where(ok.unsqueeze(0), g, torch.zeros_like(g)) + if ctx.released is not None: + rel = ctx.released[:, safe_bus] + g = torch.where(rel, torch.zeros_like(g), g) g = torch.where(torch.isnan(gen_v), torch.zeros_like(g), g) grad_gen_v = g diff --git a/src/gpusim2grid/scenario_sweep/__init__.py b/src/gpusim2grid/scenario_sweep/__init__.py index c7d1b30..e7d1d1c 100644 --- a/src/gpusim2grid/scenario_sweep/__init__.py +++ b/src/gpusim2grid/scenario_sweep/__init__.py @@ -341,6 +341,13 @@ def get_reserved_buses(self): some from set_contingency_gens' mask).""" return np.asarray(self._s.get_reserved_buses(), dtype=np.int64) + def get_row_pv_to_pq(self): + """list[list[int]]: per scenario (original row order), the AC-solver + buses the last run() turned PV->PQ because set_contingency_gens' mask + took out every generator locally regulating them. All empty without a + mask; empty before run().""" + return self._s.get_row_pv_to_pq() + def set_topology(self, branch_ids_per_scenario): """Build topology from a list-of-lists of branch indices, row-aligned with set_injections(). diff --git a/tests/python/test_batch_power_flow.py b/tests/python/test_batch_power_flow.py index 4ce60cc..afcf498 100644 --- a/tests/python/test_batch_power_flow.py +++ b/tests/python/test_batch_power_flow.py @@ -63,6 +63,35 @@ def _central_diff(f, x, idx, eps=1e-6): return (f(xp) - f(xm)) / (2 * eps) +def _fd_check(pf, load_p, load_q, gen_p, gen_v, line_status, coords, tol=1e-4, + gen_status=None): + """Central differences of a weighted |V|^2 loss vs the analytic gradient.""" + n_bus = pf.n_bus + w = torch.linspace(0.5, 1.5, n_bus, dtype=RDT, device="cuda") + + def loss_of(lp, lq, gp, gv): + V = pf(load_p=lp, load_q=lq, gen_p=gp, gen_v=gv, line_status=line_status, + gen_status=gen_status) + valid = torch.isfinite(V.real) + Vs = torch.where(valid, V, torch.zeros_like(V)) + return ((Vs.abs() ** 2) * w).sum() + + ins = {"load_p": load_p, "load_q": load_q, "gen_p": gen_p, "gen_v": gen_v} + grads = {k: v.clone().requires_grad_(True) for k, v in ins.items() if v is not None} + args = {k: grads.get(k) for k in ins} + loss_of(args["load_p"], args["load_q"], args["gen_p"], args["gen_v"]).backward() + for name, idx in coords: + def f(x, name=name): + a = {k: (v.detach() if v is not None else None) for k, v in args.items()} + a[name] = x + with torch.no_grad(): + return loss_of(a["load_p"], a["load_q"], a["gen_p"], a["gen_v"]).item() + fd = _central_diff(f, grads[name], idx) + an = grads[name].grad[idx].item() + assert abs(an - fd) <= tol * max(1.0, abs(fd)), (name, idx, an, fd) + return grads + + # --------------------------------------------------------------------------- # Forward # --------------------------------------------------------------------------- @@ -376,31 +405,9 @@ def f(lp, gp): assert torch.all(lp.grad[1] == 0) and torch.all(gp.grad[1] == 0) assert torch.any(lp.grad[0] != 0) and torch.any(lp.grad[2] != 0) - def _fd_check(self, pf, load_p, load_q, gen_p, gen_v, line_status, coords, tol=1e-4): - """Central differences of a weighted |V|^2 loss vs the analytic gradient.""" - n_bus = pf.n_bus - w = torch.linspace(0.5, 1.5, n_bus, dtype=RDT, device="cuda") + def _fd_check(self, *args, **kwargs): + return _fd_check(*args, **kwargs) - def loss_of(lp, lq, gp, gv): - V = pf(load_p=lp, load_q=lq, gen_p=gp, gen_v=gv, line_status=line_status) - valid = torch.isfinite(V.real) - Vs = torch.where(valid, V, torch.zeros_like(V)) - return ((Vs.abs() ** 2) * w).sum() - - ins = {"load_p": load_p, "load_q": load_q, "gen_p": gen_p, "gen_v": gen_v} - grads = {k: v.clone().requires_grad_(True) for k, v in ins.items() if v is not None} - args = {k: grads.get(k) for k in ins} - loss_of(args["load_p"], args["load_q"], args["gen_p"], args["gen_v"]).backward() - for name, idx in coords: - def f(x, name=name): - a = {k: (v.detach() if v is not None else None) for k, v in args.items()} - a[name] = x - with torch.no_grad(): - return loss_of(a["load_p"], a["load_q"], a["gen_p"], a["gen_v"]).item() - fd = _central_diff(f, grads[name], idx) - an = grads[name].grad[idx].item() - assert abs(an - fd) <= tol * max(1.0, abs(fd)), (name, idx, an, fd) - return grads def test_central_differences_ieee14_distributed_slack(self, ieee14_base_case): grid = ieee14_base_case["grid"] @@ -505,3 +512,322 @@ def test_matches_single_system_adjoint_on_untripped_row(self, ieee14_base_case): torch.testing.assert_close(lp.grad[0, el.load_sel], expected_lp, atol=1e-7, rtol=1e-5) torch.testing.assert_close(lq.grad[0, el.load_sel], expected_lq, atol=1e-7, rtol=1e-5) assert torch.all(lp.grad[1] == 0) # row 1 got no cotangent + + +# --------------------------------------------------------------------------- +# Call-to-call state: a call without a contingency input after one with it +# must reset the session (same inputs -> same answer, whatever ran before). +# --------------------------------------------------------------------------- + +def _two_gen_case14(): + """case14 with a 2nd generator on the bus of pp gen 2 (bus 5): the bus stays + PV while one of the two is off and turns PQ when both are. Returns + (solved grid, [gen ids on bus 5]).""" + pp = pytest.importorskip("pandapower") + import pandapower.networks as pn + from lightsim2grid.network import init_from_pandapower + from lightsim2grid.lightsim2grid_cpp import AlgorithmType + with warnings.catch_warnings(): + warnings.simplefilter("ignore") + net = pn.case14() + pp.create_gen(net, bus=5, p_mw=10.0, vm_pu=float(net.gen.vm_pu.iloc[2]), + controllable=True, min_q_mvar=-50., max_q_mvar=50.) + grid = init_from_pandapower(net) + grid.change_algorithm(AlgorithmType.NR_KLU) + n_bus = grid.get_bus_vn_kv().shape[0] + v0 = grid.dc_pf(np.ones(n_bus, dtype=complex), 1, 1e-6) + assert grid.ac_pf(v0.copy(), 30, 1e-10).shape[0] > 0 + gens = grid.get_generators() + shared = [g for g in range(len(gens)) if gens[g].bus_id == 5] + assert len(shared) == 2 + return grid, shared + + +class TestCallToCallState: + + def _loss_grad(self, pf, load_p, load_q, gen_p, **kw): + lp = load_p.clone().requires_grad_(True) + lq = load_q.clone().requires_grad_(True) + gp = gen_p.clone().requires_grad_(True) + V = pf(load_p=lp, load_q=lq, gen_p=gp, **kw) + valid = torch.isfinite(V.real) + (torch.where(valid, V, torch.zeros_like(V)).abs() ** 2).sum().backward() + return V.detach(), lp.grad, lq.grad, gp.grad + + def test_topology_is_reset_by_a_call_without_status(self, ieee14_base_case, solver_atol): + grid = ieee14_base_case["grid"] + pf, ref = _pf(grid), _pf(grid) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.05, 0.95]) + + # 1. never any topology + V0 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() + assert (pf.sweep.driver_build_counter, pf.sweep.source_build_counter) == (1, 1) + + # 2. trips on two rows (warm) + ls, ts = _all_connected(pf, n) + ls[1, 3] = False + ls[2, 0] = False + V1 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, line_status=ls, trafo_status=ts).clone() + assert (pf.sweep.driver_build_counter, pf.sweep.source_build_counter) == (1, 2) + torch.testing.assert_close(V1[0], V0[0], atol=solver_atol, rtol=0) + assert not torch.allclose(V1[1], V0[1], atol=1e-4) + assert not torch.allclose(V1[2], V0[2], atol=1e-4) + + # 3. no status inputs again: the trips must be gone (warm reset, no analysis) + V2 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() + assert (pf.sweep.driver_build_counter, pf.sweep.source_build_counter) == (1, 3) + assert pf.timings.t_analysis_ms == 0.0 + torch.testing.assert_close(V2, V0, atol=solver_atol, rtol=0) + + # 4. and stays gone on the hot path + V3 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() + assert pf.sweep.source_build_counter == 3 + torch.testing.assert_close(V3, V0, atol=solver_atol, rtol=0) + + # 5. all-True masks mean the same as no masks + ls, ts = _all_connected(pf, n) + V4 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, line_status=ls, trafo_status=ts).clone() + torch.testing.assert_close(V4, V0, atol=solver_atol, rtol=0) + + # 6. gradients after the reset match a session that never saw a trip + out = self._loss_grad(pf, load_p, load_q, gen_p) + exp = self._loss_grad(ref, load_p, load_q, gen_p) + for a, b in zip(out, exp): + torch.testing.assert_close(a, b, atol=10 * solver_atol, rtol=1e-6) + + # 7. a new row count without status after trips (the session would + # refuse a stale trip list) -> fresh answer + lp5, lq5, gp5 = _base_inputs(pf, 5, [1.0, 1.02, 0.98, 1.05, 0.95]) + ls5, ts5 = _all_connected(pf, 5) + ls5[4, 6] = False + pf(load_p=lp5, load_q=lq5, gen_p=gp5, line_status=ls5, trafo_status=ts5) + V6 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() + torch.testing.assert_close(V6, V0, atol=solver_atol, rtol=0) + + @needs_bridge + def test_gen_status_is_reset_by_a_call_without_it(self, ieee14_base_case, solver_atol): + grid = ieee14_base_case["grid"] + pf, ref = _pf(grid), _pf(grid) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.05, 0.95]) + V0 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() + assert pf.reserved_switchable_buses.size == 0 + + gs = torch.ones(n, pf.n_gen, dtype=torch.bool, device="cuda") + gs[1, 1] = False + gs[2, 2] = False + V1 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_status=gs).clone() + assert pf.reserved_switchable_buses.size == 2 # the two buses that can flip + assert pf.sweep.driver_build_counter == 2 # structure changed: cold + torch.testing.assert_close(V1[0], V0[0], atol=solver_atol, rtol=0) + assert not torch.allclose(V1[1], V0[1], atol=1e-4) + + # no gen_status: the mask is cleared and the reserved structure released + V2 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() + assert pf.reserved_switchable_buses.size == 0 + assert pf.sweep.driver_build_counter == 3 + assert pf.sweep.solver.dim_J == ref.sweep.solver.dim_J + torch.testing.assert_close(V2, V0, atol=solver_atol, rtol=0) + V3 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() # hot + assert pf.sweep.driver_build_counter == 3 + torch.testing.assert_close(V3, V0, atol=solver_atol, rtol=0) + + # all-True == no mask (no rebuild either) + V4 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, + gen_status=torch.ones_like(gs)).clone() + assert pf.sweep.driver_build_counter == 3 + torch.testing.assert_close(V4, V0, atol=solver_atol, rtol=0) + + out = self._loss_grad(pf, load_p, load_q, gen_p) + exp = self._loss_grad(ref, load_p, load_q, gen_p) + for a, b in zip(out, exp): + torch.testing.assert_close(a, b, atol=10 * solver_atol, rtol=1e-6) + + # new row count, no gen_status, after a masked call + lp5, lq5, gp5 = _base_inputs(pf, 5) + gs5 = torch.ones(5, pf.n_gen, dtype=torch.bool, device="cuda") + gs5[3, 1] = False + pf(load_p=lp5, load_q=lq5, gen_p=gp5, gen_status=gs5) + V6 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() + torch.testing.assert_close(V6, V0, atol=solver_atol, rtol=0) + + +# --------------------------------------------------------------------------- +# Generator contingencies (gen_status) +# --------------------------------------------------------------------------- + +@needs_bridge +class TestGenStatus: + + def test_forward_matches_scenario_sweep(self, ieee14_base_case, solver_atol): + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + pf = _pf(grid) + n = 4 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.05, 0.95, 1.02]) + gs = torch.ones(n, pf.n_gen, dtype=torch.bool, device="cuda") + gs[1, 1] = False + gs[2, 2] = False + gs[3, [1, 3]] = False + ls, ts = _all_connected(pf, n) + ls[2, 3] = False + gen_v = torch.full((n, pf.n_gen), float("nan"), dtype=RDT, device="cuda") + gen_v[:, 1] = 1.03 # off on rows 1 and 3: must be ignored there + gen_v[3, 2] = 1.02 + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=gen_v, + line_status=ls, trafo_status=ts, gen_status=gs) + assert torch.isfinite(V).all() + assert pf.reserved_switchable_buses.size == 3 + + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(load_p.cpu().numpy(), load_q.cpu().numpy(), + gen_p.cpu().numpy()) + sw.set_topology([[], [], [3], []]) + sw.set_contingency_gens(~gs.cpu().numpy()) + gv = gen_v.cpu().numpy().copy() + gv[~gs.cpu().numpy()] = np.nan + sw.set_gen_v(gv) + sw.compute(batch_size=n) + V_ref = sw.solver.V_results.to_numpy().reshape(n, pf.n_bus) + np.testing.assert_allclose(V.cpu().numpy(), V_ref, atol=solver_atol) + + # the disconnected generator's P is really gone: same as gen_p = 0 there + gp0 = gen_p.clone() + gp0[~gs] = 0.0 + V_gp0 = pf(load_p=load_p, load_q=load_q, gen_p=gp0, gen_v=gen_v, + line_status=ls, trafo_status=ts, gen_status=gs) + torch.testing.assert_close(V_gp0, V, atol=solver_atol, rtol=0) + + def test_forward_matches_one_off_powerflow(self, solver_atol): + """Row with both generators of a bus off (bus -> PQ) vs a fresh grid + with them deactivated, and one generator off (bus stays PV).""" + grid, (g_a, g_b) = _two_gen_case14() + pf = _pf(grid, nb_iter=15) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n) + gs = torch.ones(n, pf.n_gen, dtype=torch.bool, device="cuda") + gs[0, g_a] = False + gs[1, [g_a, g_b]] = False + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_status=gs) + assert pf.reserved_switchable_buses.tolist() == [5] + assert pf.sweep.solver.get_row_pv_to_pq() == [[], [5], []] + buses = np.asarray(grid.id_ac_solver_to_me(), dtype=int) + for row, off in enumerate([[g_a], [g_a, g_b], []]): + g2, _ = _two_gen_case14() + for g in off: + g2.deactivate_gen(int(g)) + g2.tell_solver_need_reset() + n_bus = g2.get_bus_vn_kv().shape[0] + v0 = g2.dc_pf(np.ones(n_bus, dtype=complex), 1, 1e-6) + ref = g2.ac_pf(v0.copy(), 30, 1e-10) + assert ref.shape[0] > 0 + np.testing.assert_allclose(V[row].cpu().numpy(), ref[buses], atol=10 * solver_atol) + + @fp64_only + def test_central_differences_released_bus(self, ieee14_base_case): + """Rows releasing a bus (its only generator off): the Q equation of that + bus is live (load_q there has a gradient), the off generator's gen_p / + gen_v have none, everything else matches finite differences.""" + grid = ieee14_base_case["grid"] + pf = _pf(grid, nb_iter=15) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1, 0.9]) + line_status, _ = _all_connected(pf, n) + line_status[2, 3] = False + gs = torch.ones(n, pf.n_gen, dtype=torch.bool, device="cuda") + gs[0, 1] = False + gs[2, [2, 3]] = False + gen_v = torch.tensor([[1.06, 1.045, 1.01, 1.07, 1.09]] * n, dtype=RDT, device="cuda") + bus_g1 = int(pf._gen_bus_all[1]) + loads_at_g1 = torch.nonzero(pf._load_bus_sel == bus_g1).flatten() + assert loads_at_g1.numel() == 1 # case14: one load on that bus + l1 = int(pf._load_sel[loads_at_g1[0]]) + coords = [("load_p", (0, 2)), ("load_q", (0, l1)), ("load_q", (1, l1)), + ("load_p", (2, 5)), ("load_q", (2, 4)), + ("gen_p", (0, 2)), ("gen_p", (1, 1)), ("gen_p", (2, 1)), + ("gen_p", (0, 1)), ("gen_v", (0, 1)), # off: 0 + ("gen_v", (0, 0)), ("gen_v", (0, 2)), ("gen_v", (1, 1)), + ("gen_v", (2, 1)), ("gen_v", (2, 4))] + grads = _fd_check(pf, load_p, load_q, gen_p, gen_v, line_status, coords, + gen_status=gs) + assert grads["gen_p"].grad[0, 1] == 0.0 and grads["gen_v"].grad[0, 1] == 0.0 + assert torch.all(grads["gen_p"].grad[2, [2, 3]] == 0) + assert torch.all(grads["gen_v"].grad[2, [2, 3]] == 0) + assert grads["load_q"].grad[0, l1] != 0.0 # released: Q equation live + assert grads["gen_v"].grad[1, 1] != 0.0 + + @fp64_only + def test_central_differences_pinned_reserved_bus(self): + """A reserved bus that is still PV on a row (one of its two generators + off) identity-pins its Q equation: no Q gradient there, and the + surviving generator's gen_v gradient is the plain PV one.""" + grid, (g_a, g_b) = _two_gen_case14() + pf = _pf(grid, nb_iter=15) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.05, 0.95]) + line_status, _ = _all_connected(pf, n) + gs = torch.ones(n, pf.n_gen, dtype=torch.bool, device="cuda") + gs[0, g_a] = False # bus 5 pinned (g_b keeps it PV) + gs[1, [g_a, g_b]] = False # bus 5 released + gen_v = torch.full((n, pf.n_gen), float("nan"), dtype=RDT, device="cuda") + gen_v[:, [g_a, g_b]] = 1.01 + gen_v[:, 1] = 1.045 + bus5 = int(pf._gen_bus_all[g_a]) + loads_at_5 = torch.nonzero(pf._load_bus_sel == bus5).flatten() + assert loads_at_5.numel() == 1 + l5 = int(pf._load_sel[loads_at_5[0]]) + coords = [("load_q", (0, l5)), ("load_q", (1, l5)), ("load_q", (2, l5)), + ("load_p", (0, l5)), ("load_p", (1, 3)), + ("gen_p", (0, g_b)), ("gen_p", (1, 1)), + ("gen_v", (0, g_b)), # the surviving generator of the pinned bus + ("gen_v", (0, 1)), ("gen_v", (1, 1)), + ("gen_v", (1, g_a)), ("gen_v", (1, g_b))] # released: 0 + # (no per-column FD on row 2: both generators write the same bus, the + # last column wins, so a single column's FD is 0 by construction -- + # both report the bus gradient, checked below.) + grads = _fd_check(pf, load_p, load_q, gen_p, gen_v, line_status, coords, + gen_status=gs) + assert grads["load_q"].grad[0, l5] == 0.0 # pinned: frozen Q equation + assert grads["load_q"].grad[2, l5] == 0.0 # plain PV bus + assert grads["load_q"].grad[1, l5] != 0.0 # released + assert torch.all(grads["gen_v"].grad[1, [g_a, g_b]] == 0) + assert grads["gen_v"].grad[0, g_a] == 0.0 + assert grads["gen_v"].grad[0, g_b] != 0.0 + torch.testing.assert_close(grads["gen_v"].grad[2, g_a], grads["gen_v"].grad[2, g_b]) + + @fp64_only + def test_gradcheck_with_gen_status(self, ieee14_base_case): + grid = ieee14_base_case["grid"] + pf = _pf(grid, nb_iter=15, snapshot_jacobian=True) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1, 0.9]) + gs = torch.ones(n, pf.n_gen, dtype=torch.bool, device="cuda") + gs[0, 1] = False + gs[2, 3] = False + gen_v = torch.full((n, pf.n_gen), float("nan"), dtype=RDT, device="cuda") + gen_v[:, 2] = 1.01 + gen_v[1, 1] = 1.04 + inputs = tuple(x.clone().requires_grad_(True) for x in (load_p, load_q, gen_p, gen_v)) + + def f(lp, lq, gp, gv): + V = pf(load_p=lp, load_q=lq, gen_p=gp, gen_v=gv, gen_status=gs) + return torch.view_as_real(V) + + assert torch.autograd.gradcheck(f, inputs, eps=1e-5, atol=1e-3, rtol=1e-2, + nondet_tol=1e-7) + + def test_snapshot_backward_after_mask_change_raises(self, ieee14_base_case): + grid = ieee14_base_case["grid"] + pf = _pf(grid, snapshot_jacobian=True) + n = 2 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.1]) + gs = torch.ones(n, pf.n_gen, dtype=torch.bool, device="cuda") + gs[1, 1] = False + lp = load_p.clone().requires_grad_(True) + V1 = pf(load_p=lp, load_q=load_q, gen_p=gen_p, gen_status=gs) + gs2 = gs.clone() + gs2[0, 1] = False # same reserved set, other rows pinned + pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_status=gs2) + with pytest.raises(RuntimeError, match="batch structure"): + V1.real.sum().backward() From efebaa2d67da703b6b48132fb00c27b105aac0a2 Mon Sep 17 00:00:00 2001 From: DONNOT Benjamin Date: Fri, 11 Sep 2026 19:33:14 +0200 Subject: [PATCH 3/7] add generator contingencies too Assisted-by: Claude Code (Fable 5.1) Signed-off-by: DONNOT Benjamin --- CLAUDE.md | 4 +- src/_cpp/contingency_analysis_helper.cpp | 2 + src/_cpp/contingency_analysis_helper.hpp | 8 ++ src/_cpp/ls2g_bridge.cpp | 41 ++++-- src/_cpp/python_bindings.cpp | 21 ++++ src/_cpp/scenario_sweep_session.cu | 38 +++++- src/_cpp/scenario_sweep_session.hpp | 19 +++ src/gpusim2grid/_ls2g_utils.py | 48 +++++++ src/gpusim2grid/differentiable/_batch_pf.py | 74 ++++++++++- src/gpusim2grid/injection_sweep/gpu_facade.py | 27 +++- src/gpusim2grid/scenario_sweep/__init__.py | 15 +++ src/gpusim2grid/scenario_sweep/gpu_facade.py | 32 +++++ tests/python/test_batch_power_flow.py | 48 +++++++ tests/python/test_gen_v.py | 119 ++++++++++++++++++ tests/python/test_limit_violations.py | 27 ++++ 15 files changed, 501 insertions(+), 22 deletions(-) diff --git a/CLAUDE.md b/CLAUDE.md index 687005a..f2eed6a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -120,7 +120,7 @@ Each facade is implemented in its sibling package's `gpu_facade.py` (e.g. `conti `ScenarioSweepGPU.set_contingency_gens(mask)` (`(n_scenarios, n_gen)` bool, row-aligned with `set_injections_from_elements`/`set_topology`; lightsim2grid PR #193 parity, same signature as its `ScenarioSweep.set_contingency_gens`) is the third contingency axis: per row it takes the generator's P (and its target Q if it does not regulate voltage) out of Sbus, re-weights the distributed slack without it, and — when the **last** generator locally regulating a bus is off — turns that bus PV→PQ *for that row only*. The injection side is Python's (`build_bus_injections(..., gen_off=)` in `_ls2g_utils.py`, using the new `InjectionElements.gen_target_q_mvar`/`gen_vreg_on`; the facade stores the element inputs so mask and injections may be set in either order; a caller using the per-bus `set_injections` must remove the injection itself). The labelling side is the session's: at `run()`, `ScenarioSweepSession::_prepare_gen_contingency` derives the set of buses that can flip (union over rows) **automatically from the mask** and, whenever it differs from the reserved set (`reserved_buses_`, growing *or* shrinking — an all-False mask is bit-identical to no mask), rebuilds `base_state_` once on a copy of the unextended ledger extended by `add_switchable_vm_buses` (one base-case setup + cuDSS analysis; the stored ctor inputs and `base_ledger_` exist for that rebuild, and any live DLPack view of the base buffers dies with it). Rows where a reserved bus is still PV identity-pin its Q row (`Contingency::pinned_buses`, emitted into the same identity-row stream as `handle_disconnected_grid` → `dVm = 0`, P row live); the base-case solve pins every reserved bus too (`AcPfNrState::d_pin_*`, wired into the single-system `NrIterBuffers` mask fields — the "n" case is lightsim2grid's PV labelling). Per-row slack weights (`_row_slack_weights`, mirroring `get_slack_weights_solver_without`: survivors renormalised; no participant left ⇒ the reference bus takes 1) travel as a `[n_active × n_slack]` array permuted into active-slot order like `h_Sbus_all_`, sliced per chunk, and consumed by the two MultiSlack kernels through `NrIterBuffers::slack_w_stride` (0 = shared base weights, bit-identical). Per-generator metadata comes from the bridge (`extract_gen_contingency_data`, `gen_contingency_data.hpp`; tuple mode raises). Refused like upstream: a remote controller or a generator on a group-controlled bus; `direct_base_case_factors`. `dim_J` / `reserved_switchable_buses` on the facade show the current structure. -`InjectionSweepGPU.set_gen_v(gen_v)` / `ScenarioSweepGPU.set_gen_v(gen_v)` (optional; `(n_scenarios, n_gen)` vm_pu, row-aligned with `set_injections`/`set_injections_from_elements`) mirrors lightsim2grid's own `modify_gen_v` — unlike every other `set_*`/`modify_*` setter it does **not** feed Sbus at all: a PV/slack bus's magnitude is never part of Newton-Raphson's unknown vector (bare or augmented system alike), so it is a value-only reseed of `d_V_batch` right after `tile_V` (`InjectionBatch`/`ScenarioSweepBatch::prepare_Ybus_batch`, `apply_gen_v_kernel` in `acpf_nr_kernels.cu`) — no Jacobian change, no new NR unknown. Only applied to a generator whose *own* AC-solver bus is Vm-fixed (`InjectionSweepSession`/`ScenarioSweepSession`'s `h_is_vm_fixed_bus_`, built once at construction from `pv ∪ slack_ids`) — this is what makes a disconnected, reactive-only ("PQ"), or remotely voltage-regulating (SVC / `VoltageControl`, which is PQ-classified with its own border row and therefore never in `pv`/`slack_ids`) generator's column a correct, silent no-op rather than a value that gets clobbered by the very next NR iteration or fights a fixed border-row target. NaN entries leave that `(row, gen)` untouched; left unset entirely, every row keeps the grid's own base-case voltage. The (n_gen,)→bus mapping is `InjectionElements.gen_bus` (`_ls2g_utils.py`, -1 for a disconnected generator so it can never alias a still-connected co-located one). `GenVOverride` (`contingency/gen_v_override.hpp`) is the shared host-side filtered/uploaded representation; `ScenarioSweepBatch` permutes it into active-slot order exactly like `h_Sbus_all_`. +`InjectionSweepGPU.set_gen_v(gen_v)` / `ScenarioSweepGPU.set_gen_v(gen_v)` (optional; `(n_scenarios, n_gen)` vm_pu, row-aligned with `set_injections`/`set_injections_from_elements`) mirrors lightsim2grid's own `modify_gen_v` — unlike every other `set_*`/`modify_*` setter it does **not** feed Sbus at all: a PV/slack bus's magnitude is never part of Newton-Raphson's unknown vector (bare or augmented system alike), so it is a value-only reseed of `d_V_batch` right after `tile_V` (`InjectionBatch`/`ScenarioSweepBatch::prepare_Ybus_batch`, `apply_gen_v_kernel` in `acpf_nr_kernels.cu`) — no Jacobian change, no new NR unknown. Only applied to a generator whose *own* AC-solver bus is Vm-fixed (`InjectionSweepSession`/`ScenarioSweepSession`'s `h_is_vm_fixed_bus_`, built once at construction from `pv ∪ slack_ids`) — this is what makes a disconnected, reactive-only ("PQ"), or remotely voltage-regulating (SVC / `VoltageControl`, which is PQ-classified with its own border row and therefore never in `pv`/`slack_ids`) generator's column a correct, silent no-op rather than a value that gets clobbered by the very next NR iteration or fights a fixed border-row target. NaN entries leave that `(row, gen)` untouched; left unset entirely, every row keeps the grid's own base-case voltage. The (n_gen,)→bus mapping is `InjectionElements.gen_bus` (`_ls2g_utils.py`, -1 for a disconnected generator so it can never alias a still-connected co-located one). `GenVOverride` (`contingency/gen_v_override.hpp`) is the shared host-side filtered/uploaded representation; `ScenarioSweepBatch` permutes it into active-slot order exactly like `h_Sbus_all_`. **Two connected generators on one bus with different set-points are refused, not last-writer-wins** (lightsim2grid's own `set_vm` documents last-writer-wins; |V| at a bus is unique, so no solution satisfies both): `conflicting_gen_v_rows` (`_ls2g_utils.py`, tolerance `GEN_V_CONFLICT_TOL`, only counting columns a session would apply — connected, Vm-fixed bus, non-NaN, not taken out by `set_contingency_gens`) flags the rows; `ScenarioSweepGPU.set_gen_v` and `BatchPowerFlow` (torch `scatter_reduce` version) report them NOT SIMULATED through `ScenarioSweepSession::set_skipped_rows` — a per-row `Contingency::skip` that `check_connectivity`/`compute_component_masks` turn into `disconnected` without touching the graph, so the row rides the existing compaction (NaN V/residual, `get_disconnected()` = 1, NOT_SIMULATED violation, zero gradient; a change is a warm source rebuild) — while `InjectionSweepGPU.set_gen_v` (no not-simulated concept) raises `ValueError`. ### Augmented Jacobian (ledger-driven NR) @@ -144,6 +144,8 @@ Three construction-time-only knobs select cuDSS's analysis-phase behavior, expos `ContingencyAnalysisGPU.compute_limit_violations` / `ScenarioSweepGPU.compute_limit_violations` (`contingency_analysis/_limit_violations.py`, `contingency/violation_kernels.cu`, `limit_violation_types.hpp`) fuse bus/branch limit checking into the batched GPU kernel: `ViolationElementType` (`BUS`/`LINE`/`TRAFO`/`GRID`) and `LimitViolationType` (`LOW_VOLTAGE`/`HIGH_VOLTAGE`/`CURRENT`/`NOT_SIMULATED`/`DIVERGENCE`) mirror lightsim2grid's own codes exactly (`LimitViolation.hpp`, `improve_const_ref` branch), including `GRID`/`NOT_SIMULATED`/`DIVERGENCE` — but the two `GRID` violation types are written from two different layers, not both from the kernel: `DIVERGENCE` is written by `check_limit_violations_kernel` itself, for a contingency/scenario it actually ran the solver on but whose residual is NaN or exceeds `violation_tol` (the kernel already computes this residual check as a precondition to trusting `V`, so folding it into the same compact output avoids a second round trip); `NOT_SIMULATED` is written by the Python session layer (`get_violations()`) for one the pre-check dropped before it ever reached that kernel (`BatchPfDriver`'s `d_violation_count` `-1` sentinel) — the solver was never invoked at all. gpusim2grid additionally populates `value`/`limit` with the actual residual/tol for `DIVERGENCE` entries; lightsim2grid's own convention leaves those NaN/unused for `GRID`. These codes are not pybind-bound (the kernel writes raw ints; the Python facade mirrors them as `enum.IntEnum`) — keep the three in sync by construction if you touch any of them. +Limits are pulled off the grid by `extract_limits` (`ls2g_bridge.cpp`, also behind `set_limits_from_grid`); lightsim2grid hands back an **empty** vector, not a NaN-filled one, for any container whose limits were never configured (every pandapower case without thermal limits, and `get_bus_vmin_kv` likewise), so both the bus and the per-container branch limits are padded to full size with NaN there — the session's `set_limits` rejects a branch array that is not `n_lines + n_trafos` long. + `_limit_violations.py` lives under `contingency_analysis/` for historical reasons but is workload-agnostic (`ViolationElementType`/`LimitViolationType`/`LimitViolation`/`compute_violations_n` reference no session type) — `scenario_sweep/__init__.py` imports it directly rather than duplicating it. The check itself (`check_limit_violations_kernel`, `BatchPfDriver::set_violation_limits`/`upload_branch_admittances`) is generic over any `BatchSource`, keyed only on `source_.tripped_branch_table()`; wiring it up for a new `BatchSource` is a session-level port (`compute_limit_violations_`/`violation_tol_`/`violation_capacity_`/`has_limits_`/`set_limits()`/the `get_violation_*` accessors — see `ScenarioSweepSession` for the template), not new kernel work. ### Zero-copy interop (DLPack) diff --git a/src/_cpp/contingency_analysis_helper.cpp b/src/_cpp/contingency_analysis_helper.cpp index 9bfb4fc..fd85b23 100755 --- a/src/_cpp/contingency_analysis_helper.cpp +++ b/src/_cpp/contingency_analysis_helper.cpp @@ -588,6 +588,7 @@ void check_connectivity( std::vector masked; for (auto& ctg : contingencies) { + if (ctg.skip) { ctg.disconnected = true; continue; } sc.collect_removed(ctg, values); if (fast_masked_set(ix, sc, masked)) { if (!masked.empty()) ctg.disconnected = true; @@ -618,6 +619,7 @@ void compute_component_masks( for (auto& ctg : contingencies) { ctg.masked_buses.clear(); ctg.stranded_groups.clear(); + if (ctg.skip) { ctg.disconnected = true; continue; } sc.collect_removed(ctg, values); if (!fast_masked_set(ix, sc, ctg.masked_buses)) { diff --git a/src/_cpp/contingency_analysis_helper.hpp b/src/_cpp/contingency_analysis_helper.hpp index 405f21f..61dea54 100755 --- a/src/_cpp/contingency_analysis_helper.hpp +++ b/src/_cpp/contingency_analysis_helper.hpp @@ -89,6 +89,14 @@ struct Contingency { std::vector triplets; bool disconnected = false; + // Caller-declared "not simulable" (ScenarioSweepSession::set_skipped_rows: + // e.g. two connected generators on one bus with different voltage + // set-points -- no V satisfies both). check_connectivity / + // compute_component_masks turn it into `disconnected` without looking at + // the graph, so the row is compacted out exactly like an islanded one + // (NaN voltage / residual, NOT_SIMULATED violation, disconnected flag). + bool skip = false; + // handle_disconnected_grid mode (see compute_component_masks): // masked_buses — solver bus ids OUTSIDE the largest connected component // of the patched graph (empty when the grid stays connected). diff --git a/src/_cpp/ls2g_bridge.cpp b/src/_cpp/ls2g_bridge.cpp index d2ac150..3fc8285 100644 --- a/src/_cpp/ls2g_bridge.cpp +++ b/src/_cpp/ls2g_bridge.cpp @@ -160,21 +160,39 @@ LimitData extract_limits(const ls2g::LSGrid& grid, int n_bus_solver) LimitData ld; + const eigen_real_type nan_val = std::numeric_limits::quiet_NaN(); + // Branch limits: bulk C++ accessor exists (TwoSidesContainer_rxh_A:: // get_limit_a1_ka/a2_ka) -- straight concat, same head/tail pattern as // concat_cplx above. NaN entries ("not configured") pass through as-is. + // Like the bus limits below, a container whose limits were NEVER + // configured hands back an EMPTY vector, not a NaN-filled nb() one (e.g. + // any pandapower case without thermal limits): pad it to nb() NaNs per + // container, so the concatenation is always n_lines + n_trafos long and + // lines/trafos never misalign when only one of the two is configured. { - Eigen::Ref l1_lines = lines.get_limit_a1_ka(); - Eigen::Ref l1_trafos = trafos.get_limit_a1_ka(); - ld.limit_a1_ka.resize(l1_lines.size() + l1_trafos.size()); - ld.limit_a1_ka.head(l1_lines.size()) = l1_lines; - ld.limit_a1_ka.tail(l1_trafos.size()) = l1_trafos; - - Eigen::Ref l2_lines = lines.get_limit_a2_ka(); - Eigen::Ref l2_trafos = trafos.get_limit_a2_ka(); - ld.limit_a2_ka.resize(l2_lines.size() + l2_trafos.size()); - ld.limit_a2_ka.head(l2_lines.size()) = l2_lines; - ld.limit_a2_ka.tail(l2_trafos.size()) = l2_trafos; + auto padded = [&](Eigen::Ref v, Eigen::Index nb) -> RealVect { + if (v.size() == nb) return RealVect(v); + if (v.size() != 0) + throw std::runtime_error( + "extract_limits: a branch current-limit vector has " + + std::to_string(v.size()) + " entries for " + std::to_string(nb) + + " elements"); + return RealVect::Constant(nb, nan_val); + }; + const Eigen::Index nl = static_cast(lines.nb()); + const Eigen::Index nt = static_cast(trafos.nb()); + const RealVect l1_lines = padded(lines.get_limit_a1_ka(), nl); + const RealVect l1_trafos = padded(trafos.get_limit_a1_ka(), nt); + ld.limit_a1_ka.resize(nl + nt); + ld.limit_a1_ka.head(nl) = l1_lines; + ld.limit_a1_ka.tail(nt) = l1_trafos; + + const RealVect l2_lines = padded(lines.get_limit_a2_ka(), nl); + const RealVect l2_trafos = padded(trafos.get_limit_a2_ka(), nt); + ld.limit_a2_ka.resize(nl + nt); + ld.limit_a2_ka.head(nl) = l2_lines; + ld.limit_a2_ka.tail(nt) = l2_trafos; } // Bus limits: grid.get_bus_vmin_kv()/get_bus_vmax_kv() return an EMPTY @@ -188,7 +206,6 @@ LimitData extract_limits(const ls2g::LSGrid& grid, int n_bus_solver) ls2g::RealVect vmax_model = grid.get_bus_vmax_kv(); std::vector me_to_solver = grid.id_me_to_ac_solver_numpy(); - const eigen_real_type nan_val = std::numeric_limits::quiet_NaN(); ld.bus_vmin_kv = RealVect::Constant(n_bus_solver, nan_val); ld.bus_vmax_kv = RealVect::Constant(n_bus_solver, nan_val); if (vmin_model.size() > 0) { diff --git a/src/_cpp/python_bindings.cpp b/src/_cpp/python_bindings.cpp index 4a095da..e156506 100644 --- a/src/_cpp/python_bindings.cpp +++ b/src/_cpp/python_bindings.cpp @@ -1394,6 +1394,27 @@ PYBIND11_MODULE(_gpusim2grid, m) "row-aligned with set_injections(). Requires set_branch_data() " "first. Optional: if never called, run() defaults every scenario " "to \"no branches tripped\" (a plain injection sweep).") + .def("set_skipped_rows", + [](ScenarioSweepSession& self, + pybind11::array_t mask) { + if (mask.ndim() != 1) + throw std::runtime_error( + "ScenarioSweepSession::set_skipped_rows: mask must be 1-D (n_scenarios,)"); + const bool* p = mask.data(); + std::vector m(static_cast(mask.shape(0))); + for (size_t i = 0; i < m.size(); ++i) m[i] = p[i] ? 1 : 0; + self.set_skipped_rows(m); + }, + pybind11::arg("mask"), + "(n_scenarios,) bool, row-aligned with set_injections(): True drops " + "that row from the batch as NOT SIMULATED (NaN voltage / residual, " + "disconnected flag = 1, GRID/NOT_SIMULATED violation) without " + "touching the graph -- e.g. two connected generators on one bus with " + "different voltage set-points. Takes effect on the next run() (a warm " + "source rebuild).") + .def("clear_skipped_rows", &ScenarioSweepSession::clear_skipped_rows, + "Drop any set_skipped_rows() mask.") + .def_property_readonly("has_skipped_rows", &ScenarioSweepSession::has_skipped_rows) .def("set_contingency_gens", [](ScenarioSweepSession& self, pybind11::array_t mask) { diff --git a/src/_cpp/scenario_sweep_session.cu b/src/_cpp/scenario_sweep_session.cu index a1339c4..284d578 100644 --- a/src/_cpp/scenario_sweep_session.cu +++ b/src/_cpp/scenario_sweep_session.cu @@ -583,6 +583,31 @@ void ScenarioSweepSession::set_topology( topology_dirty_ = true; } +// ============================================================================= +// set_skipped_rows / clear_skipped_rows +// ============================================================================= +void ScenarioSweepSession::set_skipped_rows(const std::vector& mask) +{ + if (mask.empty()) + throw std::runtime_error( + "ScenarioSweepSession::set_skipped_rows: n_scenarios must be > 0"); + if (has_injections_ && static_cast(mask.size()) != n_scenarios_) + throw std::runtime_error( + "ScenarioSweepSession::set_skipped_rows: row count must match " + "set_injections()'s n_scenarios"); + skip_rows_ = mask; + has_skip_ = true; + skip_dirty_ = true; +} + +void ScenarioSweepSession::clear_skipped_rows() +{ + if (!has_skip_) return; + skip_rows_.clear(); + has_skip_ = false; + skip_dirty_ = true; +} + // ============================================================================= // set_limits // ============================================================================= @@ -651,6 +676,12 @@ void ScenarioSweepSession::run() "matches set_injections()'s n_scenarios -- call set_contingency_gens() " "again after changing set_injections()"); + if (has_skip_ && static_cast(skip_rows_.size()) != n_scenarios_) + throw std::runtime_error( + "ScenarioSweepSession: set_skipped_rows()'s row count no longer " + "matches set_injections()'s n_scenarios -- call set_skipped_rows() " + "again after changing set_injections()"); + // Generator contingencies: derive what the mask needs -- the buses that // must own a reserved Vm column + Q equation (union over rows), the // per-row PV→PQ releases, the per-row slack participants taken out -- and @@ -686,11 +717,13 @@ void ScenarioSweepSession::run() // Reset disconnected flags from any previous run() — contingencies_ is // mutated in place across runs. - for (auto& ctg : contingencies_) { + for (size_t r = 0; r < contingencies_.size(); ++r) { + Contingency& ctg = contingencies_[r]; ctg.disconnected = false; ctg.masked_buses.clear(); ctg.stranded_groups.clear(); ctg.pinned_buses.clear(); + ctg.skip = has_skip_ && skip_rows_[r] != 0; } // Per-row PV pins: every reserved bus stays PV (identity Q row) except @@ -743,7 +776,7 @@ void ScenarioSweepSession::run() // ------------------------------------------------------------------------- const ScenarioSweepDriverConfig cfg = _current_config(); const bool cold = !solver_ || cfg != driver_cfg_; - const bool warm = !cold && (topology_dirty_ || gen_off_dirty_); + const bool warm = !cold && (topology_dirty_ || gen_off_dirty_ || skip_dirty_); if (cold) { // Host preprocessing (resolve_indices + connectivity/masking + @@ -892,6 +925,7 @@ void ScenarioSweepSession::run() has_violations_result_ = compute_limit_violations_; injections_dirty_ = topology_dirty_ = gen_v_dirty_ = gen_off_dirty_ = false; + skip_dirty_ = false; last_run_kept_jacobian_ = keep_final_jacobian_; ++run_counter_; diff --git a/src/_cpp/scenario_sweep_session.hpp b/src/_cpp/scenario_sweep_session.hpp index d106634..217a88a 100644 --- a/src/_cpp/scenario_sweep_session.hpp +++ b/src/_cpp/scenario_sweep_session.hpp @@ -241,6 +241,13 @@ struct ScenarioSweepSession { std::vector> tripped_branches_per_scenario_; bool has_topology_ = false; + // Caller-declared not-simulable rows (set_skipped_rows(); ORIGINAL row + // order, n_scenarios entries). Copied into Contingency::skip by run(); a + // change is a warm source rebuild (the compaction changes). + std::vector skip_rows_; + bool has_skip_ = false; + bool skip_dirty_ = false; + // ========================================================================= // Generator contingencies (set_contingency_gens(), lightsim2grid PR #193 // parity). gen_data_ is the per-generator snapshot the bridge read off the @@ -365,6 +372,18 @@ struct ScenarioSweepSession { // ========================================================================= void set_topology(const std::vector>& branch_ids_per_scenario); + // ========================================================================= + // set_skipped_rows — (n_scenarios,) bool, row-aligned with + // set_injections(): True drops that row from the batch as NOT SIMULATED + // (NaN voltage / residual, disconnected flag = 1, GRID/NOT_SIMULATED + // violation) without touching the graph. Used by the Python facades for a + // row whose connected generators on one bus carry different voltage + // set-points (no V satisfies both). clear_skipped_rows() drops the mask. + // ========================================================================= + void set_skipped_rows(const std::vector& mask); + void clear_skipped_rows(); + bool has_skipped_rows() const { return has_skip_; } + // ========================================================================= // set_gen_v -- see InjectionSweepSession::set_gen_v's doc (identical // semantics: NOT fed into Sbus, only re-seeds |V| at each generator's own diff --git a/src/gpusim2grid/_ls2g_utils.py b/src/gpusim2grid/_ls2g_utils.py index 80e2007..ca99430 100644 --- a/src/gpusim2grid/_ls2g_utils.py +++ b/src/gpusim2grid/_ls2g_utils.py @@ -344,6 +344,54 @@ def _scatter(bus, sel): load_p_base=load_p_base, load_q_base=load_q_base, gen_p_base=gen_p_base) +GEN_V_CONFLICT_TOL = 1e-8 + + +def conflicting_gen_v_rows(gen_v, gen_bus, is_vm_fixed_bus, gen_off=None, + tol=GEN_V_CONFLICT_TOL): + """Rows of a ``(n_rows, n_gen)`` ``gen_v`` matrix that ask one bus for two + different voltage magnitudes. + + Only the entries a session would actually apply count: a generator whose + own AC-solver bus is Vm-fixed (``is_vm_fixed_bus[gen_bus[g]]``), connected + (``gen_bus[g] >= 0`` and not ``gen_off[row, g]``), with a non-NaN value. + Two such generators on the same bus with set-points further apart than + ``tol`` (per-unit) make the row infeasible -- |V| at a bus is unique -- + so the callers report it as NOT SIMULATED instead of letting the last + column silently win (lightsim2grid's own ``set_vm`` behaviour). + + Returns a ``(n_rows,)`` bool array, True where the row conflicts. + """ + gen_v = np.asarray(gen_v, dtype=np.float64) + gen_bus = np.asarray(gen_bus, dtype=np.int64) + fixed = np.asarray(is_vm_fixed_bus, dtype=bool) + n_rows, n_gen = gen_v.shape + ok = (gen_bus >= 0) + ok[ok] = fixed[gen_bus[ok]] + cols = np.flatnonzero(ok) + out = np.zeros(n_rows, dtype=bool) + if cols.size < 2: + return out + bus = gen_bus[cols] + # only buses with at least two candidate columns can conflict + _, inv, cnt = np.unique(bus, return_inverse=True, return_counts=True) + multi = cnt[inv] > 1 + if not multi.any(): + return out + cols, bus = cols[multi], bus[multi] + vals = gen_v[:, cols] + active = np.isfinite(vals) + if gen_off is not None: + active &= ~np.asarray(gen_off, dtype=bool)[:, cols] + for b in np.unique(bus): + sel = bus == b + v = np.where(active[:, sel], vals[:, sel], np.nan) + with np.errstate(invalid="ignore"): + spread = np.nanmax(v, axis=1) - np.nanmin(v, axis=1) + out |= np.nan_to_num(spread, nan=0.0) > tol + return out + + def build_bus_injections(elements, load_p, load_q, gen_p, gen_off=None): """Assemble per-bus (p_mw, q_mvar) from per-element injection matrices. diff --git a/src/gpusim2grid/differentiable/_batch_pf.py b/src/gpusim2grid/differentiable/_batch_pf.py index 019b709..0a50e79 100644 --- a/src/gpusim2grid/differentiable/_batch_pf.py +++ b/src/gpusim2grid/differentiable/_batch_pf.py @@ -26,6 +26,15 @@ ``gen_v`` (through the seeded voltage magnitude). ``line_status`` / ``trafo_status`` / ``gen_status`` are discrete and carry no gradient. +A row asking one bus for two different magnitudes -- two connected +generators on that bus, both with a non-NaN ``gen_v``, set-points further +apart than ``gen_v_conflict_tol`` -- is infeasible (|V| at a bus is unique) +and comes back NOT SIMULATED like an islanded row: NaN voltage, zero +gradient, ``get_disconnected()`` = 1 (``ScenarioSweepGPU`` reports the same +through ``set_skipped_rows``; lightsim2grid's own ``set_vm`` would silently +let the last column win). Co-located generators with the *same* set-point +are fine and both report the bus gradient. + Generator contingencies (``gen_status``) are lightsim2grid's ``ScenarioSweep.set_contingency_gens`` (see ``ScenarioSweepGPU``): a disconnected generator's P leaves Sbus (and the target Q of a non @@ -116,7 +125,7 @@ from torch import Tensor from .. import _gpusim2grid as _cpp -from .._ls2g_utils import extract_branch_data +from .._ls2g_utils import extract_branch_data, GEN_V_CONFLICT_TOL from ..scenario_sweep.gpu_facade import ScenarioSweepGPU from ._flows import compute_flows as _compute_flows_torch @@ -131,7 +140,8 @@ class BatchPowerFlow: :meth:`forward`). """ - def __init__(self, sweep, *, snapshot_jacobian=False): + def __init__(self, sweep, *, snapshot_jacobian=False, + gen_v_conflict_tol=GEN_V_CONFLICT_TOL): if sweep._elements is None: raise ValueError( "BatchPowerFlow needs a ScenarioSweepGPU built from a lightsim2grid " @@ -140,6 +150,7 @@ def __init__(self, sweep, *, snapshot_jacobian=False): self._solver = sweep.solver # _ScenarioSweepSolver self._solver.fixed_batch_capacity = True # always one chunk (adjoint) self.snapshot_jacobian = bool(snapshot_jacobian) + self.gen_v_conflict_tol = float(gen_v_conflict_tol) self._dev = torch.device("cuda", _device_index(sweep)) self._rdtype = torch.float32 if bool(_cpp.is_fp32) else torch.float64 @@ -203,6 +214,19 @@ def __init__(self, sweep, *, snapshot_jacobian=False): self._gen_off_in_session = False self._pending_gen_off = None # mask to hand to the session on the next run self._released = None # (n_scen, n_bus) bool of the last run, or None + self._skip_in_session = False + self._pending_skip = None # (n_scen,) bool ndarray, or "clear" + # Columns that can conflict: connected generators on a Vm-fixed bus + # sharing that bus with another such generator. + gb = np.asarray(el.gen_bus, dtype=np.int64) + fixed = np.asarray(self._solver.is_vm_fixed_bus, dtype=bool) + ok = gb >= 0 + ok[ok] = fixed[gb[ok]] + _, inv, cnt = np.unique(gb[ok], return_inverse=True, return_counts=True) + cols = np.flatnonzero(ok)[cnt[inv] > 1] + self._shared_cols = torch.as_tensor(cols, device=dev) if cols.size else None + if self._shared_cols is not None: + self._shared_bus = torch.as_tensor(gb[cols], device=dev) # ------------------------------------------------------------------ build @classmethod @@ -212,12 +236,13 @@ def from_lsgrid(cls, grid, *, nb_iter=4, handle_disconnected_grid=False, pivot_epsilon_alg=None, use_distributed_slack=True, scaling_max_voltage_change=None, max_dVa=None, max_dVm=None, init_from_n_powerflow=True, max_iter_base=10, tol_base=1e-8, - precision=None): + precision=None, gen_v_conflict_tol=GEN_V_CONFLICT_TOL): """Build from a *solved* lightsim2grid grid (``grid.ac_pf`` done). The keyword arguments are :class:`gpusim2grid.ScenarioSweepGPU`'s - (same meaning), plus ``strategy`` (linear-solve strategy string) and - ``snapshot_jacobian`` (see the module docstring). ``nb_iter`` is the + (same meaning), plus ``strategy`` (linear-solve strategy string), + ``snapshot_jacobian`` and ``gen_v_conflict_tol`` (see the module + docstring). ``nb_iter`` is the fixed Newton-Raphson iteration count per row: raise it (and lower ``tol_base``) when gradients must be accurate -- the adjoint is exact only at a converged solution. @@ -232,7 +257,8 @@ def from_lsgrid(cls, grid, *, nb_iter=4, handle_disconnected_grid=False, max_dVa=max_dVa, max_dVm=max_dVm, use_distributed_slack=use_distributed_slack) sweep.strategy = strategy - return cls(sweep, snapshot_jacobian=snapshot_jacobian) + return cls(sweep, snapshot_jacobian=snapshot_jacobian, + gen_v_conflict_tol=gen_v_conflict_tol) # --------------------------------------------------------------- forward def __call__(self, *args, **kwargs): @@ -288,6 +314,7 @@ def forward(self, load_p=None, load_q=None, gen_p=None, gen_v=None, Q = Q.index_add(1, self._gen_bus_sel, -q_off[:, self._gen_sel] * inv_sn) if gen_v is not None: gen_v = torch.where(gen_off, torch.full_like(gen_v, float("nan")), gen_v) + self._apply_gen_v_conflicts(gen_v, n_scen) if self._gen_sel.numel(): P = P.index_add(1, self._gen_bus_sel, gen_p[:, self._gen_sel] * inv_sn) if self._load_sel.numel(): @@ -420,6 +447,33 @@ def _apply_topology(self, line_status, trafo_status, n_scen): self._pending_topology = ragged self._topology_mask = tripped.clone() + def _apply_gen_v_conflicts(self, gen_v, n_scen): + """Rows asking one bus for two different |V| (see the module docstring) + are handed to the session as skipped rows; the session mask is cleared + again once no row conflicts. Same pending pattern as the topology.""" + self._pending_skip = None + bad = None + if gen_v is not None and self._shared_cols is not None: + v = gen_v.detach()[:, self._shared_cols] + finite = torch.isfinite(v) + big, small = torch.finfo(v.dtype).max, -torch.finfo(v.dtype).max + hi = torch.full((n_scen, self.n_bus), small, dtype=v.dtype, device=v.device) + lo = torch.full((n_scen, self.n_bus), big, dtype=v.dtype, device=v.device) + idx = self._shared_bus.unsqueeze(0).expand(n_scen, -1) + hi = hi.scatter_reduce(1, idx, torch.where(finite, v, torch.full_like(v, small)), + reduce="amax") + lo = lo.scatter_reduce(1, idx, torch.where(finite, v, torch.full_like(v, big)), + reduce="amin") + both = (hi > small) & (lo < big) + conflict = both & ((hi - lo) > self.gen_v_conflict_tol) + rows = conflict.any(dim=1) + if bool(rows.any()): + bad = rows.cpu().numpy() + if bad is not None: + self._pending_skip = bad + elif self._skip_in_session: + self._pending_skip = "clear" + def _apply_gen_status(self, gen_status, n_scen): """Same contract as _apply_topology for the generator mask. Returns the (n_scen, n_gen) bool "disconnected" mask to apply to the injections @@ -473,6 +527,14 @@ def forward(ctx, P: Tensor, Q: Tensor, gen_v, pf: BatchPowerFlow) -> Tensor: pf._sweep.set_topology(pf._pending_topology) pf._topology_in_session = True pf._pending_topology = None + if pf._pending_skip is not None: + if isinstance(pf._pending_skip, str): + solver.clear_skipped_rows() + pf._skip_in_session = False + else: + solver.set_skipped_rows(pf._pending_skip) + pf._skip_in_session = True + pf._pending_skip = None if pf._pending_gen_off is not None: # After the injections: the session checks the row counts against # them. The facade validates the mask (and would re-assemble Sbus diff --git a/src/gpusim2grid/injection_sweep/gpu_facade.py b/src/gpusim2grid/injection_sweep/gpu_facade.py index c3dfd71..784070e 100644 --- a/src/gpusim2grid/injection_sweep/gpu_facade.py +++ b/src/gpusim2grid/injection_sweep/gpu_facade.py @@ -9,6 +9,8 @@ It is a thin facade over :class:`_InjectionSweepSolver`. """ +import numpy as np + from . import ( _InjectionSweepSolver, _normalize_device, @@ -21,6 +23,7 @@ extract_branch_data, extract_injection_elements, build_bus_injections, + conflicting_gen_v_rows, grid_from_pandapower, _validate_precision, ) @@ -254,6 +257,13 @@ def __init__(self, grid, *, init_from_n_powerflow=True, precision="fp64", self._elements = None # no loads/generators to read else: self._elements = extract_injection_elements(grid, self._inner.n_bus) + # |V|-fixed buses (pv ∪ slack, AC-solver numbering) -- the only + # buses set_gen_v() acts on; used to spot two connected generators + # asking one bus for two different magnitudes. + fixed = np.zeros(self._inner.n_bus, dtype=bool) + fixed[np.asarray(grid.get_pv(), dtype=np.int64)] = True + fixed[np.asarray(grid.get_slack_ids(), dtype=np.int64)] = True + self._is_vm_fixed_bus = fixed self._init_from_n_powerflow = bool(init_from_n_powerflow) self._last_residuals = None @@ -345,7 +355,12 @@ def set_gen_v(self, gen_v): silently ignored, mirroring lightsim2grid's own ``voltage_regulator_on_``-gated behavior. Left unset entirely (the default), every scenario keeps the grid's own base-case - voltage. + voltage. Raises ``ValueError`` when a row asks one bus for two + different magnitudes (two connected generators on that bus with + set-points further apart than + ``gpusim2grid._ls2g_utils.GEN_V_CONFLICT_TOL``): no V satisfies + both, and lightsim2grid's own ``set_vm`` would silently let the + last one win. Notes ----- @@ -357,6 +372,16 @@ def set_gen_v(self, gen_v): raise RuntimeError( "set_gen_v() needs a lightsim2grid grid; explicit-array " "(tuple) mode has no generators to read.") + gen_v = np.ascontiguousarray(gen_v, dtype=np.float64) + bad = conflicting_gen_v_rows(gen_v, self._elements.gen_bus, self._is_vm_fixed_bus) + if bad.any(): + rows = np.flatnonzero(bad) + raise ValueError( + f"set_gen_v: rows {rows[:10].tolist()}{'...' if rows.size > 10 else ''} " + "ask one bus for two different voltage magnitudes (two connected " + "generators on the same bus with different set-points): no V " + "satisfies both. Give co-located generators the same vm_pu, or " + "NaN for all but one of them.") self._inner.set_gen_v(gen_v, self._elements.gen_bus) def compute(self, batch_size=512): diff --git a/src/gpusim2grid/scenario_sweep/__init__.py b/src/gpusim2grid/scenario_sweep/__init__.py index e7d1d1c..6bad721 100644 --- a/src/gpusim2grid/scenario_sweep/__init__.py +++ b/src/gpusim2grid/scenario_sweep/__init__.py @@ -341,6 +341,21 @@ def get_reserved_buses(self): some from set_contingency_gens' mask).""" return np.asarray(self._s.get_reserved_buses(), dtype=np.int64) + def set_skipped_rows(self, mask): + """(n_scenarios,) bool, row-aligned with set_injections(): True drops + that row as NOT SIMULATED (NaN voltage / residual, disconnected flag + = 1, GRID/NOT_SIMULATED violation) without touching the graph. Takes + effect on the next run() (a warm source rebuild).""" + self._s.set_skipped_rows(np.ascontiguousarray(mask, dtype=bool)) + + def clear_skipped_rows(self): + """Drop any set_skipped_rows() mask.""" + self._s.clear_skipped_rows() + + @property + def has_skipped_rows(self): + return self._s.has_skipped_rows + def get_row_pv_to_pq(self): """list[list[int]]: per scenario (original row order), the AC-solver buses the last run() turned PV->PQ because set_contingency_gens' mask diff --git a/src/gpusim2grid/scenario_sweep/gpu_facade.py b/src/gpusim2grid/scenario_sweep/gpu_facade.py index a67e944..9306ec7 100644 --- a/src/gpusim2grid/scenario_sweep/gpu_facade.py +++ b/src/gpusim2grid/scenario_sweep/gpu_facade.py @@ -39,6 +39,7 @@ extract_branch_data, extract_injection_elements, build_bus_injections, + conflicting_gen_v_rows, grid_from_pandapower, _validate_precision, ) @@ -253,6 +254,9 @@ def __init__(self, grid, *, init_from_n_powerflow=True, precision="fp64", # disconnected generators taken out, whatever the call order. self._pending_elements = None self._gen_off = None + # set_gen_v() input, kept so a later set_contingency_gens() (or vice + # versa) can re-derive which rows ask one bus for two different |V|. + self._gen_v = None # ------------------------------------------------------------------ spec def set_branch_data(self, branch_from, branch_to, yff_eff, yft_eff, ytf_eff, ytt_eff, @@ -367,6 +371,7 @@ def set_contingency_gens(self, mask): self._gen_off = mask if self._pending_elements is not None: self._assemble_injections() + self._update_skipped_rows() @property def dim_J(self): @@ -406,6 +411,15 @@ def set_gen_v(self, gen_v): (the default), every scenario keeps the grid's own base-case voltage. + A row asking one bus for two different magnitudes (two connected + generators on that bus, both applied, set-points further apart than + ``gpusim2grid._ls2g_utils.GEN_V_CONFLICT_TOL``) is infeasible -- |V| + at a bus is unique -- and is reported as NOT SIMULATED (NaN voltage / + residual, :meth:`get_disconnected` = 1, a ``GRID``/``NOT_SIMULATED`` + violation) rather than letting the last column silently win the way + lightsim2grid's own ``set_vm`` does. A generator taken out by + :meth:`set_contingency_gens` (or a NaN entry) does not take part. + Notes ----- The generator -> bus wiring is snapshotted at construction, like @@ -416,7 +430,25 @@ def set_gen_v(self, gen_v): raise RuntimeError( "set_gen_v() needs a lightsim2grid grid; explicit-array " "(tuple) mode has no generators to read.") + gen_v = np.ascontiguousarray(gen_v, dtype=np.float64) self._inner.set_gen_v(gen_v, self._elements.gen_bus) + self._gen_v = gen_v + self._update_skipped_rows() + + def _update_skipped_rows(self): + """Re-derive the not-simulable rows from set_gen_v() (and the + generator mask); a row-count mismatch is left to the C++ session.""" + if self._gen_v is None: + return + gen_off = self._gen_off + if gen_off is not None and gen_off.shape[0] != self._gen_v.shape[0]: + gen_off = None + bad = conflicting_gen_v_rows(self._gen_v, self._elements.gen_bus, + self._inner.is_vm_fixed_bus, gen_off=gen_off) + if bad.any(): + self._inner.set_skipped_rows(bad) + else: + self._inner.clear_skipped_rows() def set_topology(self, branch_ids_per_scenario): """Define each scenario's topology as branch removals. diff --git a/tests/python/test_batch_power_flow.py b/tests/python/test_batch_power_flow.py index afcf498..0489a5d 100644 --- a/tests/python/test_batch_power_flow.py +++ b/tests/python/test_batch_power_flow.py @@ -831,3 +831,51 @@ def test_snapshot_backward_after_mask_change_raises(self, ieee14_base_case): pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_status=gs2) with pytest.raises(RuntimeError, match="batch structure"): V1.real.sum().backward() + + +class TestConflictingSetpoints: + """Two connected generators on one bus with different gen_v: infeasible, + the row comes back NOT SIMULATED (NaN, zero gradient), and is simulated + again once the set-points agree, one is NaN, or one is disconnected.""" + + def test_row_is_dropped_and_recovers(self, solver_atol): + grid, (g_a, g_b) = _two_gen_case14() + pf = _pf(grid, nb_iter=15) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.05, 0.95]) + gen_v = torch.full((n, pf.n_gen), float("nan"), dtype=RDT, device="cuda") + gen_v[:, [g_a, g_b]] = 1.02 + V_ok = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=gen_v).clone() + assert torch.isfinite(V_ok).all() + + bad = gen_v.clone() + bad[1, g_b] = 1.03 + gv = bad.clone().requires_grad_(True) + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=gv) + assert pf.get_disconnected().tolist() == [0, 1, 0] + assert torch.isnan(V[1]).all() + torch.testing.assert_close(V[[0, 2]], V_ok[[0, 2]], atol=solver_atol, rtol=0) + loss = torch.where(torch.isfinite(V.real), V, torch.zeros_like(V)).abs().pow(2).sum() + loss.backward() + assert torch.all(gv.grad[1] == 0) + assert gv.grad[0, g_a] == gv.grad[0, g_b] != 0 + + # NaN on one of them: applied set-point unique again + one = bad.clone() + one[1, g_a] = float("nan") + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=one) + assert pf.get_disconnected().tolist() == [0, 0, 0] + assert torch.isfinite(V).all() + assert abs(V[1, int(pf._gen_bus_all[g_b])]).item() == pytest.approx(1.03, abs=1e-6) + + # disconnecting the conflicting generator resolves it too + gs = torch.ones(n, pf.n_gen, dtype=torch.bool, device="cuda") + gs[1, g_b] = False + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=bad, gen_status=gs) + assert pf.get_disconnected().tolist() == [0, 0, 0] + assert abs(V[1, int(pf._gen_bus_all[g_a])]).item() == pytest.approx(1.02, abs=1e-6) + + # and back to the agreeing set-points: identical to the first call + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, gen_v=gen_v) + assert pf.get_disconnected().tolist() == [0, 0, 0] + torch.testing.assert_close(V, V_ok, atol=solver_atol, rtol=0) diff --git a/tests/python/test_gen_v.py b/tests/python/test_gen_v.py index 70baaba..64c9704 100644 --- a/tests/python/test_gen_v.py +++ b/tests/python/test_gen_v.py @@ -204,3 +204,122 @@ def test_tuple_mode_rejects_set_gen_v(self, ieee14_base_case): nb_iter=4, init_from_n_powerflow=False) with pytest.raises(RuntimeError): isw.set_gen_v(np.zeros((1, 1))) + + +# --------------------------------------------------------------------------- +# Two connected generators on one bus with different set-points: |V| at a +# bus is unique, so no solution satisfies both. lightsim2grid's set_vm lets +# the last one silently win; gpusim2grid refuses the row instead. +# --------------------------------------------------------------------------- + +def _two_gen_case14(): + """case14 with a 2nd generator on the bus of pp gen 2 (bus 5). Returns + (solved grid, [gen ids on bus 5], solver bus id of bus 5).""" + pp = pytest.importorskip("pandapower") + import warnings + import pandapower.networks as pn + from lightsim2grid.network import init_from_pandapower + from lightsim2grid.lightsim2grid_cpp import AlgorithmType + with warnings.catch_warnings(): + warnings.simplefilter("ignore") + net = pn.case14() + pp.create_gen(net, bus=5, p_mw=10.0, vm_pu=float(net.gen.vm_pu.iloc[2]), + controllable=True, min_q_mvar=-50., max_q_mvar=50.) + grid = init_from_pandapower(net) + grid.change_algorithm(AlgorithmType.NR_KLU) + n_bus = grid.get_bus_vn_kv().shape[0] + v0 = grid.dc_pf(np.ones(n_bus, dtype=complex), 1, 1e-6) + assert grid.ac_pf(v0.copy(), 30, 1e-10).shape[0] > 0 + gens = grid.get_generators() + shared = [g for g in range(len(gens)) if gens[g].bus_id == 5] + assert len(shared) == 2 + buses = np.asarray(grid.id_ac_solver_to_me(), dtype=int) + return grid, shared, int(np.flatnonzero(buses == 5)[0]) + + +def test_conflicting_gen_v_rows_helper(): + from gpusim2grid._ls2g_utils import conflicting_gen_v_rows + gen_bus = np.array([0, 3, 3, 3, -1, 5]) # gens 1,2,3 share bus 3; gen 4 off + fixed = np.array([True, False, False, True, False, False]) # bus 5 is PQ + nan = float("nan") + gen_v = np.array([ + [1.0, 1.02, 1.02, nan, 1.5, 1.1], # equal set-points -> ok + [1.0, 1.02, 1.03, nan, 1.5, 1.1], # 1.02 vs 1.03 on bus 3 -> conflict + [1.0, 1.02, nan, 1.04, 1.5, 1.1], # 1.02 vs 1.04 -> conflict + [1.0, nan, nan, 1.04, 1.5, 1.1], # one applied -> ok + [1.0, 1.02, 1.02 + 1e-10, nan, 1.5, 1.1], # within tolerance -> ok + ]) + np.testing.assert_array_equal( + conflicting_gen_v_rows(gen_v, gen_bus, fixed), + [False, True, True, False, False]) + # a generator taken out by a contingency does not take part + gen_off = np.zeros(gen_v.shape, dtype=bool) + gen_off[1, 2] = True + np.testing.assert_array_equal( + conflicting_gen_v_rows(gen_v, gen_bus, fixed, gen_off=gen_off), + [False, False, True, False, False]) + + +@requires_gpu +class TestConflictingSetpoints: + def test_scenario_sweep_row_is_not_simulated(self, solver_atol): + from gpusim2grid import ScenarioSweepGPU + from gpusim2grid.contingency_analysis._limit_violations import ( + LimitViolationType, ViolationElementType) + grid, (g_a, g_b), b5 = _two_gen_case14() + n_bus = grid.get_Ybus_solver().shape[0] + load_p, load_q = grid.get_loads_res_full()[:2] + gen_p = np.asarray(grid.get_gen_target_p()) + n = 3 + rep = lambda a: np.repeat(np.asarray(a)[None, :], n, axis=0) # noqa: E731 + n_gen = len(grid.get_generators()) + gen_v = np.full((n, n_gen), np.nan) + gen_v[:, [g_a, g_b]] = 1.02 + gen_v[1, g_b] = 1.03 # row 1: infeasible + + sw = ScenarioSweepGPU(grid, nb_iter=10, tol_base=1e-10) + sw.set_injections_from_elements(rep(load_p), rep(load_q), rep(gen_p)) + sw.set_gen_v(gen_v) + sw.set_limits_from_grid() # case14 has no limits: all NaN + sw.compute_limit_violations = True + sw.compute(batch_size=n) + V = sw.solver.V_results.to_numpy().reshape(n, n_bus) + assert sw.get_disconnected().tolist() == [0, 1, 0] + viol = sw.get_violations() + assert viol[0] == [] and viol[2] == [] + assert len(viol[1]) == 1 and viol[1][0].element_type == ViolationElementType.GRID + assert viol[1][0].violation_type == LimitViolationType.NOT_SIMULATED + assert np.all(np.isnan(V[1])) and np.isnan(sw.last_residuals()[1]) + assert np.all(np.isfinite(V[[0, 2]])) + np.testing.assert_allclose(np.abs(V[0, b5]), 1.02, atol=solver_atol) + + # same set-points again -> the row is back (warm reset of the skip) + gen_v[1, g_b] = 1.02 + sw.set_gen_v(gen_v) + sw.compute(batch_size=n) + assert sw.get_disconnected().tolist() == [0, 0, 0] + V2 = sw.solver.V_results.to_numpy().reshape(n, n_bus) + np.testing.assert_allclose(V2[1], V[0], atol=solver_atol) + + # the conflict disappears when one of the two is disconnected + gen_v[1, g_b] = 1.03 + mask = np.zeros((n, n_gen), dtype=bool) + mask[1, g_b] = True + sw.set_gen_v(gen_v) + sw.set_contingency_gens(mask) + sw.compute(batch_size=n) + assert sw.get_disconnected().tolist() == [0, 0, 0] + V3 = sw.solver.V_results.to_numpy().reshape(n, n_bus) + np.testing.assert_allclose(np.abs(V3[1, b5]), 1.02, atol=solver_atol) + + def test_injection_sweep_raises(self): + from gpusim2grid import InjectionSweepGPU + grid, (g_a, g_b), _ = _two_gen_case14() + n_gen = len(grid.get_generators()) + gen_v = np.full((2, n_gen), np.nan) + gen_v[:, [g_a, g_b]] = 1.02 + isw = InjectionSweepGPU(grid, nb_iter=8, tol_base=1e-10) + isw.set_gen_v(gen_v) # equal: fine + gen_v[1, g_a] = 1.0 + with pytest.raises(ValueError, match=r"rows \[1\]"): + isw.set_gen_v(gen_v) diff --git a/tests/python/test_limit_violations.py b/tests/python/test_limit_violations.py index eed4fdc..b358b3a 100644 --- a/tests/python/test_limit_violations.py +++ b/tests/python/test_limit_violations.py @@ -438,3 +438,30 @@ def test_not_simulated_contingency_reports_grid_entry(): assert v.violation_type == LimitViolationType.NOT_SIMULATED assert v.element_id == -1 assert np.isnan(v.value) and np.isnan(v.limit) + + +@requires_gpu +def test_set_limits_from_grid_without_configured_limits(ieee14_base_case): + """A grid whose current limits were never configured (lightsim2grid then + returns EMPTY limit vectors, not NaN-filled ones) must still give + n_lines + n_trafos NaN entries, so set_limits_from_grid() works and the + fused check reports no violation.""" + from gpusim2grid import ScenarioSweepGPU, ContingencyAnalysisGPU + grid = ieee14_base_case["grid"] + n_branch = len(grid.get_lines()) + len(grid.get_trafos()) + sw = ScenarioSweepGPU(grid, nb_iter=6, tol_base=1e-10) + _, _, a1, a2, n_lines = sw._extract_limits_arrays() + assert a1.shape == (n_branch,) and a2.shape == (n_branch,) + assert np.all(np.isnan(a1)) and np.all(np.isnan(a2)) + assert n_lines == len(grid.get_lines()) + sw.set_limits_from_grid() + sw.compute_limit_violations = True + load_p, load_q = grid.get_loads_res_full()[:2] + gen_p = np.asarray(grid.get_gen_target_p()) + rep = lambda a: np.repeat(np.asarray(a)[None, :], 2, axis=0) # noqa: E731 + sw.set_injections_from_elements(rep(load_p), rep(load_q), rep(gen_p)) + sw.compute(batch_size=2) + assert sw.get_violations() == [[], []] + # the constructor-time path (bridge factory) must accept it too + ca = ContingencyAnalysisGPU(grid, nb_iter=6, tol_base=1e-10, compute_limit_violations=True) + assert ca is not None From e38fa0b6e1355ee88fbb9f5fdc3e0b0cff43ecee Mon Sep 17 00:00:00 2001 From: DONNOT Benjamin Date: Fri, 11 Sep 2026 21:12:19 +0200 Subject: [PATCH 4/7] fix a bug: missing a nan propagation in the divergence flag / f inf kernel Assisted-by: Claude Code (Opus 5) Signed-off-by: DONNOT Benjamin --- src/_cpp/acpf_nr_kernels.cu | 20 +++++++++++++------- tests/python/test_injection_batch.py | 28 ++++++++++++++++++++++++++++ 2 files changed, 41 insertions(+), 7 deletions(-) diff --git a/src/_cpp/acpf_nr_kernels.cu b/src/_cpp/acpf_nr_kernels.cu index 5e10b4f..a16abeb 100755 --- a/src/_cpp/acpf_nr_kernels.cu +++ b/src/_cpp/acpf_nr_kernels.cu @@ -949,21 +949,27 @@ __global__ void compute_residuals_kernel( const cuda_real_type* F_b = d_F + b * dim_J; cuda_real_type local_max = cuda_real_type(0); - // Each thread scans its portion of F_b. + // Each thread scans its portion of F_b. NaN must PROPAGATE: a plain + // `v > local_max` is false for NaN, so a slot whose F is entirely NaN + // (e.g. a NaN cuDSS solve poisoning V) would otherwise report residual 0 + // and look converged. `nan_max` is a sticky-NaN max: once any element of + // F_b is NaN the slot's residual is NaN. (The lambda is a plain __device__ + // helper; keeping it local avoids adding a header symbol for one use.) + auto nan_max = [](cuda_real_type a, cuda_real_type b) -> cuda_real_type { + return (isnan(a) || isnan(b)) ? cuda_real_type(NAN) : (b > a ? b : a); + }; for (int i = threadIdx.x; i < dim_J; i += blockDim.x) { cuda_real_type v = F_b[i]; if (v < cuda_real_type(0)) v = -v; - if (v > local_max) local_max = v; + local_max = nan_max(local_max, v); } sdata[threadIdx.x] = local_max; __syncthreads(); - // Tree reduction within the block. + // Tree reduction within the block (NaN-propagating, see above). for (int stride = blockDim.x / 2; stride > 0; stride >>= 1) { - if (threadIdx.x < stride) { - if (sdata[threadIdx.x + stride] > sdata[threadIdx.x]) - sdata[threadIdx.x] = sdata[threadIdx.x + stride]; - } + if (threadIdx.x < stride) + sdata[threadIdx.x] = nan_max(sdata[threadIdx.x], sdata[threadIdx.x + stride]); __syncthreads(); } diff --git a/tests/python/test_injection_batch.py b/tests/python/test_injection_batch.py index e7e9017..98e97c3 100644 --- a/tests/python/test_injection_batch.py +++ b/tests/python/test_injection_batch.py @@ -444,6 +444,34 @@ def _make_solver(self, ieee14_base_case, batch_size=5, nb_iter=10): max_iter_base=10, tol_base=1e-6, ) + def test_nan_scenario_reports_nan_residual(self, ieee14_base_case, residual_atol): + """A scenario whose NR state is NaN must report a NaN residual, not 0. + + Regression for ``compute_residuals_kernel``: its ‖F‖∞ reduction used a + plain ``v > local_max`` compare, which is false for NaN, so a slot whose + F was entirely NaN (e.g. after a NaN cuDSS solve poisoned V) reported + residual 0 and looked converged. A NaN injection is the simplest way to + force such a slot deterministically; the neighbouring scenarios must + still solve and report finite residuals. + """ + scales = [0.9, 1.0, 1.1] + p_mw, q_mvar, sn_mva = _build_scenarios(ieee14_base_case, scales) + p_mw = p_mw.copy() + p_mw[1, :] = np.nan + + solver = self._make_solver(ieee14_base_case, batch_size=3, nb_iter=10) + solver.set_injections(p_mw, q_mvar, sn_mva) + solver.run() + V = solver.V_results.to_numpy().reshape(3, ieee14_base_case["n_bus"]) + res = solver.residuals.to_numpy() + + assert np.isnan(V[1]).any(), "the NaN injection did not poison scenario 1's V" + assert np.isnan(res[1]), ( + f"scenario 1 has NaN voltages but residual {res[1]!r} (NaN dropped by the max reduction)") + for s in (0, 2): + assert np.all(np.isfinite(V[s])) + assert np.isfinite(res[s]) and res[s] < residual_atol + def test_run_matches_one_shot(self, ieee14_base_case): """Stateful solver result must match the one-shot acpf_nr_gpu_injection.""" scales = [0.9, 1.0, 1.1] From cc28dd51325e97ba2f6e1d0b4f941bc4fd2044b6 Mon Sep 17 00:00:00 2001 From: Benjamin Donnot Date: Tue, 29 Sep 2026 17:18:03 +0000 Subject: [PATCH 5/7] fix scenario-sweep state kept across reused drivers Three bugs in the persistent-driver path of PR #14, each with a regression test: - ScenarioSweepSession::run(): a hot run (only injections / gen_v changed) keeps the live batch source but used to clear every row's disconnected / masked_buses flags, which only a new source recomputes. Islanded rows still came back NaN while get_disconnected() and n_disconnected reported nothing. The path is now chosen first and the flags are reset only when a new source is built. - ScenarioSweepSession::set_branch_data(): the live driver uploads the branch admittances / flow buffers once, so data set after the first run never reached the GPU (wrong flows, and an out-of-bounds write in zero_branch_flows_kernel if the branch count grew). It now invalidates the driver's cached branch data so the next run / compute_flows re-uploads. - BatchPowerFlow: the topology and generator masks were cached before the session received them; a call raising in between (e.g. a bad gen_status shape) left the cache ahead of the session, and the next call with the same line_status was treated as unchanged and solved on the old trips. The masks are now committed only once the session accepted them. Assisted-by: Claude Code Claude-Session: https://claude.ai/code/session_01FZwHuL9k42bdoU93GqaPQW Signed-off-by: Benjamin Donnot --- src/_cpp/scenario_sweep_session.cu | 54 +++++++++++------ src/gpusim2grid/differentiable/_batch_pf.py | 18 ++++-- tests/python/test_batch_power_flow.py | 29 ++++++++++ .../python/test_scenario_sweep_persistence.py | 58 +++++++++++++++++++ 4 files changed, 137 insertions(+), 22 deletions(-) diff --git a/src/_cpp/scenario_sweep_session.cu b/src/_cpp/scenario_sweep_session.cu index 284d578..22da532 100644 --- a/src/_cpp/scenario_sweep_session.cu +++ b/src/_cpp/scenario_sweep_session.cu @@ -488,6 +488,16 @@ void ScenarioSweepSession::set_branch_data( h_bus_vn_kv_ = bus_vn_kv; sn_mva_ = sn_mva; has_branch_data_ = true; + + // A live driver caches the admittances / flow buffers it uploaded (once + // per driver, see compute_flows() and run()). Invalidate them so the new + // data -- possibly a different branch count -- is re-uploaded; otherwise + // flows / limit violations keep using the old admittances, and a new + // tripped-branch id >= the old count indexes past the flow buffers. + if (solver_) { + solver_->_has_branch_admittances = false; + solver_->_has_branch_data = false; + } } // ============================================================================= @@ -708,7 +718,19 @@ void ScenarioSweepSession::run() "'direct_refactor_every_n'."); } - if (!has_topology_) { + // ------------------------------------------------------------------------- + // Cold / warm / hot path selection against the live driver. Decided + // before touching contingencies_: only a cold or warm run builds a new + // batch source, which is what (re)computes the per-row disconnected / + // masked_buses flags. A hot run keeps the live source, so it must also + // keep those flags (get_disconnected(), n_disconnected) as they are. + // ------------------------------------------------------------------------- + const ScenarioSweepDriverConfig cfg = _current_config(); + const bool cold = !solver_ || cfg != driver_cfg_; + const bool warm = !cold && (topology_dirty_ || gen_off_dirty_ || skip_dirty_); + const bool new_source = cold || warm; + + if (new_source && !has_topology_) { // Default: no branches tripped for any scenario (pure injection sweep). contingencies_.assign(static_cast(n_scenarios_), Contingency{}); tripped_branches_per_scenario_.assign( @@ -716,19 +738,22 @@ void ScenarioSweepSession::run() } // Reset disconnected flags from any previous run() — contingencies_ is - // mutated in place across runs. - for (size_t r = 0; r < contingencies_.size(); ++r) { - Contingency& ctg = contingencies_[r]; - ctg.disconnected = false; - ctg.masked_buses.clear(); - ctg.stranded_groups.clear(); - ctg.pinned_buses.clear(); - ctg.skip = has_skip_ && skip_rows_[r] != 0; + // mutated in place across runs, and the new source recomputes them. + if (new_source) { + for (size_t r = 0; r < contingencies_.size(); ++r) { + Contingency& ctg = contingencies_[r]; + ctg.disconnected = false; + ctg.masked_buses.clear(); + ctg.stranded_groups.clear(); + ctg.pinned_buses.clear(); + ctg.skip = has_skip_ && skip_rows_[r] != 0; + } } // Per-row PV pins: every reserved bus stays PV (identity Q row) except - // the ones this row turned PQ. Nothing reserved → nothing to pin. - if (!reserved_buses_.empty()) { + // the ones this row turned PQ. Nothing reserved → nothing to pin. (A hot + // run has the same mask, hence the same pins, as the live source.) + if (new_source && !reserved_buses_.empty()) { for (int r = 0; r < n_scenarios_; ++r) { const std::vector& to_pq = row_pv_to_pq[static_cast(r)]; std::vector& pinned = contingencies_[static_cast(r)].pinned_buses; @@ -771,13 +796,6 @@ void ScenarioSweepSession::run() "ScenarioSweepSession: injection buffer size does not match " "n_scenarios x n_bus (call set_injections() again)"); - // ------------------------------------------------------------------------- - // Cold / warm / hot path selection against the live driver. - // ------------------------------------------------------------------------- - const ScenarioSweepDriverConfig cfg = _current_config(); - const bool cold = !solver_ || cfg != driver_cfg_; - const bool warm = !cold && (topology_dirty_ || gen_off_dirty_ || skip_dirty_); - if (cold) { // Host preprocessing (resolve_indices + connectivity/masking + // build_flat_patches), mutates contingencies_ in-place so the diff --git a/src/gpusim2grid/differentiable/_batch_pf.py b/src/gpusim2grid/differentiable/_batch_pf.py index 0a50e79..0a52223 100644 --- a/src/gpusim2grid/differentiable/_batch_pf.py +++ b/src/gpusim2grid/differentiable/_batch_pf.py @@ -206,13 +206,19 @@ def __init__(self, sweep, *, snapshot_jacobian=False, # Call-to-call state. self._last_n_scen = None + # The *_mask attributes describe what the session actually holds: they + # are only updated by the op once the session accepted the pending + # value, so a call that raises in between cannot leave them claiming + # a topology / generator mask the session never received. self._topology_mask = None # (n_scen, n_branch) bool, True = tripped self._topology_in_session = False self._pending_topology = None # ragged list to hand to the session on the next run + self._pending_topology_mask = None # _topology_mask once _pending_topology is set self._gen_v_in_session = False self._gen_off_mask = None # (n_scen, n_gen) bool, True = disconnected self._gen_off_in_session = False self._pending_gen_off = None # mask to hand to the session on the next run + self._pending_gen_off_mask = None # _gen_off_mask once _pending_gen_off is set self._released = None # (n_scen, n_bus) bool of the last run, or None self._skip_in_session = False self._pending_skip = None # (n_scen,) bool ndarray, or "clear" @@ -418,13 +424,13 @@ def _apply_topology(self, line_status, trafo_status, n_scen): ragged list (if any) is handed over by the op AFTER the injections (the session checks the row counts against them).""" self._pending_topology = None + self._pending_topology_mask = None if line_status is None and trafo_status is None: # No trips wanted. Only touch the session if it still holds trips # or its row count is stale. if self._topology_mask is not None or ( self._topology_in_session and n_scen != self._last_n_scen): self._pending_topology = [[] for _ in range(n_scen)] - self._topology_mask = None return tripped = torch.cat([ @@ -445,7 +451,7 @@ def _apply_topology(self, line_status, trafo_status, n_scen): else: ragged = [[] for _ in range(n_scen)] self._pending_topology = ragged - self._topology_mask = tripped.clone() + self._pending_topology_mask = tripped.clone() def _apply_gen_v_conflicts(self, gen_v, n_scen): """Rows asking one bus for two different |V| (see the module docstring) @@ -482,11 +488,11 @@ def _apply_gen_status(self, gen_status, n_scen): all-False mask over (bit-identical to no mask for the session, which also drops the reserved structure it no longer needs).""" self._pending_gen_off = None + self._pending_gen_off_mask = None if gen_status is None: if self._gen_off_mask is not None or ( self._gen_off_in_session and n_scen != self._last_n_scen): self._pending_gen_off = np.zeros((n_scen, self.n_gen), dtype=bool) - self._gen_off_mask = None return None gen_off = ~self._as_status(gen_status, self.n_gen, n_scen, "gen_status") @@ -498,7 +504,7 @@ def _apply_gen_status(self, gen_status, n_scen): prev = self._gen_off_mask if prev is None or prev.shape != gen_off.shape or not torch.equal(prev, gen_off): self._pending_gen_off = np.ascontiguousarray(gen_off.cpu().numpy(), dtype=bool) - self._gen_off_mask = gen_off.clone() + self._pending_gen_off_mask = gen_off.clone() return gen_off if bool(gen_off.any()) else None @@ -525,8 +531,10 @@ def forward(ctx, P: Tensor, Q: Tensor, gen_v, pf: BatchPowerFlow) -> Tensor: solver.set_injections_dlpack(S.__dlpack__(), stream) if pf._pending_topology is not None: pf._sweep.set_topology(pf._pending_topology) + pf._topology_mask = pf._pending_topology_mask pf._topology_in_session = True pf._pending_topology = None + pf._pending_topology_mask = None if pf._pending_skip is not None: if isinstance(pf._pending_skip, str): solver.clear_skipped_rows() @@ -541,8 +549,10 @@ def forward(ctx, P: Tensor, Q: Tensor, gen_v, pf: BatchPowerFlow) -> Tensor: # only for set_injections_from_elements inputs, which this path # never uses -- the injections above already carry the correction). pf._sweep.set_contingency_gens(pf._pending_gen_off) + pf._gen_off_mask = pf._pending_gen_off_mask pf._gen_off_in_session = True pf._pending_gen_off = None + pf._pending_gen_off_mask = None if gen_v is not None: solver.set_gen_v_dlpack(gen_v.detach().contiguous().__dlpack__(), pf._gen_bus_np, stream) pf._gen_v_in_session = True diff --git a/tests/python/test_batch_power_flow.py b/tests/python/test_batch_power_flow.py index 0489a5d..73539b0 100644 --- a/tests/python/test_batch_power_flow.py +++ b/tests/python/test_batch_power_flow.py @@ -652,6 +652,35 @@ def test_gen_status_is_reset_by_a_call_without_it(self, ieee14_base_case, solver V6 = pf(load_p=load_p, load_q=load_q, gen_p=gen_p).clone() torch.testing.assert_close(V6, V0, atol=solver_atol, rtol=0) + def test_failed_call_does_not_desync_topology(self, ieee14_base_case, solver_atol): + # A call that raises after the topology was decided (here: a bad + # gen_status shape, validated after line_status) must not leave the + # cached mask claiming a topology the session never received -- + # otherwise the next call with that line_status looks "unchanged" and + # silently solves on the old trips. + grid = ieee14_base_case["grid"] + pf, ref = _pf(grid), _pf(grid) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.05, 0.95]) + ls, ts = _all_connected(pf, n) + ls[1, 3] = False + pf(load_p=load_p, load_q=load_q, gen_p=gen_p, line_status=ls, trafo_status=ts) + assert pf.sweep.source_build_counter == 1 + + ls2 = ls.clone() + ls2[2, 0] = False + bad_gs = torch.ones(n, pf.n_gen + 1, dtype=torch.bool, device="cuda") + with pytest.raises(ValueError, match="gen_status"): + pf(load_p=load_p, load_q=load_q, gen_p=gen_p, + line_status=ls2, trafo_status=ts, gen_status=bad_gs) + + V = pf(load_p=load_p, load_q=load_q, gen_p=gen_p, + line_status=ls2, trafo_status=ts).clone() + assert pf.sweep.source_build_counter == 2 # new topology: warm + V_ref = ref(load_p=load_p, load_q=load_q, gen_p=gen_p, + line_status=ls2, trafo_status=ts) + torch.testing.assert_close(V, V_ref, atol=solver_atol, rtol=0) + # --------------------------------------------------------------------------- # Generator contingencies (gen_status) diff --git a/tests/python/test_scenario_sweep_persistence.py b/tests/python/test_scenario_sweep_persistence.py index 1e391f0..730db7c 100644 --- a/tests/python/test_scenario_sweep_persistence.py +++ b/tests/python/test_scenario_sweep_persistence.py @@ -349,3 +349,61 @@ def test_set_injections_dlpack_matches_numpy(self, ieee14_base_case, solver_atol wrong_dtype = torch.zeros(3, sw.n_bus, dtype=torch.float64, device="cuda") with pytest.raises(RuntimeError, match="dtype"): sw.solver.set_injections_dlpack(wrong_dtype.__dlpack__()) + + +class TestStateKeptAcrossRuns: + """Session state that must survive (or be refreshed by) a reused driver.""" + + @needs_bridge + @pytest.mark.parametrize("change", ["injections", "gen_v"]) + def test_disconnected_flags_survive_a_hot_run(self, change): + # Only a new batch source recomputes which rows are islanded; a hot + # run keeps the live source, so it must keep those flags too. + from gpusim2grid import ScenarioSweepGPU + grid, _, spur_line, _ = _solved_spur_grid(distributed_slack=False) + scales = [1.0, 1.05, 0.95] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(*_rows(grid, scales)) + sw.set_topology([[], [int(spur_line)], []]) # row 1 islands the spur bus + V1 = _np(sw.compute(batch_size=3)) + assert list(sw.get_disconnected()) == [0, 1, 0] + assert sw.timings.n_disconnected == 1 + assert np.all(np.isnan(V1[1])) + + if change == "injections": + sw.set_injections_from_elements(*_rows(grid, [0.9, 1.1, 1.02])) + else: + gen_v = np.full((3, len(grid.get_gen_target_p())), np.nan) + gen_v[:, 0] = 1.02 + sw.set_gen_v(gen_v) + V2 = _np(sw.compute(batch_size=3)) + assert sw.driver_build_counter == 1 and sw.source_build_counter == 1 # hot + assert np.all(np.isnan(V2[1])) + assert list(sw.get_disconnected()) == [0, 1, 0] + assert sw.timings.n_disconnected == 1 + + def test_set_branch_data_after_a_run_reaches_the_driver(self, ieee14_base_case, solver_atol): + # The driver uploads the branch admittances once and survives across + # runs: a later set_branch_data() must still reach it. + from gpusim2grid import ScenarioSweepGPU + from gpusim2grid._ls2g_utils import extract_branch_data + grid = ieee14_base_case["grid"] + scales = [1.0, 1.1] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(*_rows(grid, scales)) + sw.compute(batch_size=2) + sw.compute_flows() + or1 = sw.or_amps.to_numpy().copy() + ex1 = sw.ex_amps.to_numpy().copy() + assert np.all(np.isfinite(or1)) and np.any(or1 > 0.0) + + # Branch currents are linear in the admittances; Ybus (hence V) is + # untouched by set_branch_data, so doubling them doubles the currents. + (b_from, b_to, yff, yft, ytf, ytt, vn_kv, sn_mva), _, _ = extract_branch_data(grid) + sw.set_branch_data(b_from, b_to, 2 * np.asarray(yff), 2 * np.asarray(yft), + 2 * np.asarray(ytf), 2 * np.asarray(ytt), vn_kv, sn_mva) + sw.compute(batch_size=2) + assert sw.driver_build_counter == 1 # same driver + sw.compute_flows() + np.testing.assert_allclose(sw.or_amps.to_numpy(), 2 * or1, rtol=solver_atol, atol=solver_atol) + np.testing.assert_allclose(sw.ex_amps.to_numpy(), 2 * ex1, rtol=solver_atol, atol=solver_atol) From d0856f1da55381ff7f2dc8e242f2289ec4645d41 Mon Sep 17 00:00:00 2001 From: Benjamin Donnot Date: Tue, 29 Sep 2026 17:58:16 +0000 Subject: [PATCH 6/7] fix five more issues in the reused scenario-sweep driver Found by a second review of PR #14 (including cc28dd5); each fix comes with a regression test. - BatchPowerFlow: the "session still holds a trip list / generator mask for another row count" check compared against _last_n_scen, which is updated before the op runs. After a call that failed inside the op, the next call with the same row count sent nothing and every run() was then refused ("row count no longer matches"). The row count of what the session holds is now recorded when the session accepts it. - ScenarioSweepSession::run(): a warm run kept the chunk capacity the cold run sized for its own active rows (1 when all but one row was islanded), so a later warm run with every row active was solved in one-row chunks. The driver is now rebuilt when the live capacity would need more than twice a fresh driver's chunk count; the old driver is freed before the new one is allocated. - compute_limit_violations = False had no effect on a reused driver: the fused check stayed armed with the old limits. run() now disarms it. - set_branch_data(): refuses data that drops a branch the current topology trips (before replacing anything), rebuilds the per-row Ybus patches from the new admittances, and drops per-branch limits of a different length so compute_limit_violations asks for set_limits() again instead of reading past their end. - differentiable compute_flows(): a half-open branch's open end (bus -1) read V[-1] and vn_kv[-1], i.e. the last bus. It now follows compute_branch_flows_kernel: V = 0 and no terminal current on that side, base current from the live endpoint. Assisted-by: Claude Code Claude-Session: https://claude.ai/code/session_01FZwHuL9k42bdoU93GqaPQW Signed-off-by: Benjamin Donnot --- src/_cpp/scenario_sweep_session.cu | 136 +++++++++++++----- src/gpusim2grid/differentiable/_batch_pf.py | 8 +- src/gpusim2grid/differentiable/_flows.py | 25 +++- tests/python/test_batch_power_flow.py | 30 ++++ tests/python/test_diff_flows.py | 56 ++++++++ .../python/test_scenario_sweep_persistence.py | 114 +++++++++++++++ 6 files changed, 325 insertions(+), 44 deletions(-) diff --git a/src/_cpp/scenario_sweep_session.cu b/src/_cpp/scenario_sweep_session.cu index 22da532..db18c71 100644 --- a/src/_cpp/scenario_sweep_session.cu +++ b/src/_cpp/scenario_sweep_session.cu @@ -479,6 +479,20 @@ void ScenarioSweepSession::set_branch_data( Eigen::Ref bus_vn_kv, double sn_mva) { + // The trip lists refer to branch ids: refuse data that no longer holds + // one of them before anything is replaced (the session stays usable). + const int n_bra_new = static_cast(branch_from.size()); + if (has_topology_) { + for (const auto& ids : tripped_branches_per_scenario_) + for (int id : ids) + if (id >= n_bra_new) + throw std::runtime_error( + "ScenarioSweepSession::set_branch_data: the current topology " + "trips branch " + std::to_string(id) + ", out of range for " + "the new " + std::to_string(n_bra_new) + " branches -- call " + "set_topology() with valid ids first"); + } + h_branch_from_ = branch_from; h_branch_to_ = branch_to; h_yff_eff_ = yff_eff; @@ -489,6 +503,24 @@ void ScenarioSweepSession::set_branch_data( sn_mva_ = sn_mva; has_branch_data_ = true; + // Branch limits are per branch: a different count makes them stale, so + // compute_limit_violations asks for set_limits() again instead of + // reading past their end. + if (has_limits_ && h_branch_limit_a1_ka_.size() != n_bra_new) { + has_limits_ = false; + has_violations_result_ = false; + } + + // The per-row Ybus patches were built from the previous admittances: + // rebuild them from the same trip lists (also marks the topology dirty, + // so the next run() builds a new source). A trip list whose row count no + // longer matches the injections is left as is: run() refuses it anyway. + if (has_topology_ && (!has_injections_ || + static_cast(tripped_branches_per_scenario_.size()) == n_scenarios_)) { + const std::vector> trips = tripped_branches_per_scenario_; + set_topology(trips); + } + // A live driver caches the admittances / flow buffers it uploaded (once // per driver, see compute_flows() and run()). Invalidate them so the new // data -- possibly a different branch count -- is re-uploaded; otherwise @@ -737,9 +769,14 @@ void ScenarioSweepSession::run() static_cast(n_scenarios_), std::vector{}); } - // Reset disconnected flags from any previous run() — contingencies_ is - // mutated in place across runs, and the new source recomputes them. - if (new_source) { + // Per-row state a new batch source computes from scratch: reset the + // disconnected flags from any previous run() (contingencies_ is mutated + // in place across runs, and the connectivity checks only ever set them) + // and re-derive the per-row PV pins -- every reserved bus stays PV + // (identity Q row) except the ones this row turned PQ; nothing reserved + // → nothing to pin. A hot run has the same mask, hence the same pins, as + // the live source, so it leaves all of this alone. + auto reset_rows = [&]() { for (size_t r = 0; r < contingencies_.size(); ++r) { Contingency& ctg = contingencies_[r]; ctg.disconnected = false; @@ -748,19 +785,16 @@ void ScenarioSweepSession::run() ctg.pinned_buses.clear(); ctg.skip = has_skip_ && skip_rows_[r] != 0; } - } - - // Per-row PV pins: every reserved bus stays PV (identity Q row) except - // the ones this row turned PQ. Nothing reserved → nothing to pin. (A hot - // run has the same mask, hence the same pins, as the live source.) - if (new_source && !reserved_buses_.empty()) { - for (int r = 0; r < n_scenarios_; ++r) { - const std::vector& to_pq = row_pv_to_pq[static_cast(r)]; - std::vector& pinned = contingencies_[static_cast(r)].pinned_buses; - for (int b : reserved_buses_) - if (!std::binary_search(to_pq.begin(), to_pq.end(), b)) pinned.push_back(b); + if (!reserved_buses_.empty()) { + for (int r = 0; r < n_scenarios_; ++r) { + const std::vector& to_pq = row_pv_to_pq[static_cast(r)]; + std::vector& pinned = contingencies_[static_cast(r)].pinned_buses; + for (int b : reserved_buses_) + if (!std::binary_search(to_pq.begin(), to_pq.end(), b)) pinned.push_back(b); + } } - } + }; + if (new_source) reset_rows(); std::vector h_slack_w_rows = _row_slack_weights(row_slack_off); const int n_bus = base_state_->n_bus; @@ -796,7 +830,49 @@ void ScenarioSweepSession::run() "ScenarioSweepSession: injection buffer size does not match " "n_scenarios x n_bus (call set_injections() again)"); - if (cold) { + // A warm run keeps the live driver's chunk capacity, which the cold run + // sized for ITS active rows. If this topology leaves many more rows + // active (e.g. the cold run had most rows islanded or skipped), that + // capacity would split them into far more chunks than a fresh driver + // would use -- rebuild the driver instead once it would take more than + // twice a fresh driver's chunk count (a small growth is cheaper served by + // one extra chunk than by a new cuDSS analysis). fixed_batch_capacity + // keeps batch_size verbatim, so it never needs this. + bool build_driver = cold; + if (warm) { + // New topology / generator mask on the live driver: rebuild only the + // source (CPU connectivity + patches + H→D of those) at the driver's + // capacity; no analysis, no allocation of the chunk buffers. + const int live_capacity = solver_->batch_size_; + ScenarioSweepBatch source( + contingencies_, + Ybus_rm_.outerIndexPtr(), + Ybus_rm_.innerIndexPtr(), + Ybus_rm_, + batch_size_, + mask_cfg_, + handle_disconnected_grid_, + GenVOverride{}, + std::vector(h_slack_w_rows), + base_state_->n_slack, + /*forced_batch_size=*/live_capacity); + int n_active = 0; + for (const auto& ctg : contingencies_) + if (!ctg.disconnected) ++n_active; + const int chunks_live = (n_active + live_capacity - 1) / live_capacity; + const int chunks_fresh = (n_active + batch_size_ - 1) / batch_size_; + if (!fixed_batch_capacity_ && chunks_live > 2 * chunks_fresh) { + build_driver = true; + reset_rows(); // the cold source below recomputes them + } else { + solver_->replace_source(std::move(source)); + used_batch_size_ = live_capacity; + solver_->mark_reused(/*hot=*/false); + ++source_build_counter_; + } + } + + if (build_driver) { // Host preprocessing (resolve_indices + connectivity/masking + // build_flat_patches), mutates contingencies_ in-place so the // disconnected flags are observable below. gen_v is set afterwards @@ -816,6 +892,8 @@ void ScenarioSweepSession::run() used_batch_size_ = source.used_batch_size(); // New driver: allocation, block-diagonal structure, cuDSS ANALYSIS. + // Free the old one first so the two never coexist on the device. + solver_.reset(); solver_ = std::make_unique( *base_state_, std::move(source), @@ -833,27 +911,7 @@ void ScenarioSweepSession::run() driver_cfg_ = cfg; ++driver_build_counter_; ++source_build_counter_; - } else if (warm) { - // New topology / generator mask on the live driver: rebuild only the - // source (CPU connectivity + patches + H→D of those) at the driver's - // capacity; no analysis, no allocation of the chunk buffers. - ScenarioSweepBatch source( - contingencies_, - Ybus_rm_.outerIndexPtr(), - Ybus_rm_.innerIndexPtr(), - Ybus_rm_, - batch_size_, - mask_cfg_, - handle_disconnected_grid_, - GenVOverride{}, - std::move(h_slack_w_rows), - base_state_->n_slack, - /*forced_batch_size=*/solver_->batch_size_); - solver_->replace_source(std::move(source)); - used_batch_size_ = solver_->batch_size_; - solver_->mark_reused(/*hot=*/false); - ++source_build_counter_; - } else { + } else if (!warm) { // Hot: only injections / gen_v changed (or nothing at all). solver_->mark_reused(/*hot=*/true); } @@ -911,6 +969,10 @@ void ScenarioSweepSession::run() h_branch_limit_a1_ka_, h_branch_limit_a2_ka_, violation_tol_, violation_capacity_, n_lines_); t_limits_setup_ms = solver_->violation_setup_ms(); + } else { + // The driver outlives the flag: a previous run's set_violation_limits + // left the fused check armed, so disarm it explicitly. + solver_->_fused_violations_enabled = false; } timings_ = solver_->solve(); diff --git a/src/gpusim2grid/differentiable/_batch_pf.py b/src/gpusim2grid/differentiable/_batch_pf.py index 0a52223..2f1ba2d 100644 --- a/src/gpusim2grid/differentiable/_batch_pf.py +++ b/src/gpusim2grid/differentiable/_batch_pf.py @@ -212,11 +212,13 @@ def __init__(self, sweep, *, snapshot_jacobian=False, # a topology / generator mask the session never received. self._topology_mask = None # (n_scen, n_branch) bool, True = tripped self._topology_in_session = False + self._topology_rows = None # row count of the trip list the session holds self._pending_topology = None # ragged list to hand to the session on the next run self._pending_topology_mask = None # _topology_mask once _pending_topology is set self._gen_v_in_session = False self._gen_off_mask = None # (n_scen, n_gen) bool, True = disconnected self._gen_off_in_session = False + self._gen_off_rows = None # row count of the mask the session holds self._pending_gen_off = None # mask to hand to the session on the next run self._pending_gen_off_mask = None # _gen_off_mask once _pending_gen_off is set self._released = None # (n_scen, n_bus) bool of the last run, or None @@ -429,7 +431,7 @@ def _apply_topology(self, line_status, trafo_status, n_scen): # No trips wanted. Only touch the session if it still holds trips # or its row count is stale. if self._topology_mask is not None or ( - self._topology_in_session and n_scen != self._last_n_scen): + self._topology_in_session and n_scen != self._topology_rows): self._pending_topology = [[] for _ in range(n_scen)] return @@ -491,7 +493,7 @@ def _apply_gen_status(self, gen_status, n_scen): self._pending_gen_off_mask = None if gen_status is None: if self._gen_off_mask is not None or ( - self._gen_off_in_session and n_scen != self._last_n_scen): + self._gen_off_in_session and n_scen != self._gen_off_rows): self._pending_gen_off = np.zeros((n_scen, self.n_gen), dtype=bool) return None @@ -532,6 +534,7 @@ def forward(ctx, P: Tensor, Q: Tensor, gen_v, pf: BatchPowerFlow) -> Tensor: if pf._pending_topology is not None: pf._sweep.set_topology(pf._pending_topology) pf._topology_mask = pf._pending_topology_mask + pf._topology_rows = len(pf._pending_topology) pf._topology_in_session = True pf._pending_topology = None pf._pending_topology_mask = None @@ -550,6 +553,7 @@ def forward(ctx, P: Tensor, Q: Tensor, gen_v, pf: BatchPowerFlow) -> Tensor: # never uses -- the injections above already carry the correction). pf._sweep.set_contingency_gens(pf._pending_gen_off) pf._gen_off_mask = pf._pending_gen_off_mask + pf._gen_off_rows = int(pf._pending_gen_off.shape[0]) pf._gen_off_in_session = True pf._pending_gen_off = None pf._pending_gen_off_mask = None diff --git a/src/gpusim2grid/differentiable/_flows.py b/src/gpusim2grid/differentiable/_flows.py index d279205..cc33ed7 100644 --- a/src/gpusim2grid/differentiable/_flows.py +++ b/src/gpusim2grid/differentiable/_flows.py @@ -14,6 +14,9 @@ base_A = sn_mva * 1e6 / (sqrt(3) * vn_kv[from] * 1e3) +(a side at bus -1 -- Kron-reduced half-open end -- counts as V = 0 with no +terminal current, and base_A then uses vn_kv[to]). + All operations are natively differentiable via PyTorch autograd. ``V`` may carry any number of leading batch dimensions (``(..., n_bus)``, e.g. the ``(n_scen, n_bus)`` output of ``BatchPowerFlow``): the branch indexing is @@ -50,16 +53,28 @@ def compute_flows( All values are real tensors of shape [..., n_branches] (the leading dimensions of ``V``). """ - Vi = V[..., branch_from] # complex [..., n_branches] - Vj = V[..., branch_to] # complex [..., n_branches] + # A side lightsim2grid Kron-reduced away (half-open line, isolated bus) + # is relabeled to bus -1: it has no voltage (V = 0) and no terminal, so no + # current or power on that side, and the base current uses the live + # endpoint's nominal voltage -- exactly compute_branch_flows_kernel. + # (Plain V[..., -1] would silently read the last bus instead.) + live_f = branch_from >= 0 + live_t = branch_to >= 0 + bf = branch_from.clamp(min=0) + bt = branch_to.clamp(min=0) + zero = torch.zeros((), dtype=V.dtype, device=V.device) + + Vi = torch.where(live_f, V[..., bf], zero) # complex [..., n_branches] + Vj = torch.where(live_t, V[..., bt], zero) # complex [..., n_branches] - I_or = yff_eff * Vi + yft_eff * Vj # origin terminal current - I_ex = ytf_eff * Vi + ytt_eff * Vj # extremity terminal current + I_or = torch.where(live_f, yff_eff * Vi + yft_eff * Vj, zero) # origin terminal current + I_ex = torch.where(live_t, ytf_eff * Vi + ytt_eff * Vj, zero) # extremity terminal current S_or = Vi * I_or.conj() # complex apparent power (pu), origin S_ex = Vj * I_ex.conj() # complex apparent power (pu), extremity - base_A = sn_mva * 1e6 / (math.sqrt(3.0) * bus_vn_kv[branch_from] * 1e3) + vn_kv = bus_vn_kv[torch.where(live_f, bf, bt)] + base_A = sn_mva * 1e6 / (math.sqrt(3.0) * vn_kv * 1e3) return { "p_or_mw": S_or.real * sn_mva, diff --git a/tests/python/test_batch_power_flow.py b/tests/python/test_batch_power_flow.py index 73539b0..48fe2d5 100644 --- a/tests/python/test_batch_power_flow.py +++ b/tests/python/test_batch_power_flow.py @@ -681,6 +681,36 @@ def test_failed_call_does_not_desync_topology(self, ieee14_base_case, solver_ato line_status=ls2, trafo_status=ts) torch.testing.assert_close(V, V_ref, atol=solver_atol, rtol=0) + def test_failed_call_with_a_new_row_count_does_not_stick(self, ieee14_base_case, + solver_atol, monkeypatch): + # The session holds an (all-empty) trip list for 3 rows. A 5-row call + # fails inside the op after the injections were handed over; the next + # 5-row call without status must still replace that 3-row list rather + # than trust a row count the session never received (it would refuse + # every such call with "row count no longer matches"). + grid = ieee14_base_case["grid"] + pf, ref = _pf(grid), _pf(grid) + lp3, lq3, gp3 = _base_inputs(pf, 3) + ls3, ts3 = _all_connected(pf, 3) + ls3[1, 3] = False + pf(load_p=lp3, load_q=lq3, gen_p=gp3, line_status=ls3, trafo_status=ts3) + pf(load_p=lp3, load_q=lq3, gen_p=gp3) # trips cleared, 3 rows + + lp5, lq5, gp5 = _base_inputs(pf, 5, [1.0, 1.02, 0.98, 1.05, 0.95]) + ls5, ts5 = _all_connected(pf, 5) + ls5[4, 6] = False + + def boom(*args, **kwargs): + raise RuntimeError("injected failure") + with monkeypatch.context() as m: + m.setattr(pf.sweep, "set_topology", boom) + with pytest.raises(RuntimeError, match="injected failure"): + pf(load_p=lp5, load_q=lq5, gen_p=gp5, line_status=ls5, trafo_status=ts5) + + V = pf(load_p=lp5, load_q=lq5, gen_p=gp5).clone() + V_ref = ref(load_p=lp5, load_q=lq5, gen_p=gp5) + torch.testing.assert_close(V, V_ref, atol=solver_atol, rtol=0) + # --------------------------------------------------------------------------- # Generator contingencies (gen_status) diff --git a/tests/python/test_diff_flows.py b/tests/python/test_diff_flows.py index 3f0fd9f..05f494a 100644 --- a/tests/python/test_diff_flows.py +++ b/tests/python/test_diff_flows.py @@ -205,3 +205,59 @@ def loss_fn(Sr, Si): nondet_tol=1e-6, # CUDA scatter_add has float rounding non-determinism ) assert result + + +class TestHalfOpenEndpoint: + + def test_minus_one_endpoint_matches_the_kernel_convention(self): + """A side at bus -1 (Kron-reduced half-open end) must not index V[-1]. + + Same convention as compute_branch_flows_kernel: that side has V = 0 + and no terminal current, and the base current uses the live + endpoint's nominal voltage. The last bus is given a different voltage + and nominal kV so that reading it by mistake changes every output. + """ + from gpusim2grid.differentiable import compute_flows + + cdt, rdt, dev = torch.complex128, torch.float64, "cuda" + V = torch.tensor([[1.0 + 0.0j, 0.98 - 0.05j, 1.02 + 0.01j, 0.7 + 0.3j], + [1.0 + 0.0j, 0.97 - 0.06j, 1.01 + 0.02j, 0.6 + 0.2j]], + dtype=cdt, device=dev, requires_grad=True) + vn_kv = torch.tensor([138.0, 138.0, 20.0, 400.0], dtype=rdt, device=dev) + # branch 0: regular (0 -> 1); branch 1: open "to" end (2 -> -1); + # branch 2: open "from" end (-1 -> 1) + b_from = torch.tensor([0, 2, -1], device=dev) + b_to = torch.tensor([1, -1, 1], device=dev) + yff = torch.tensor([1 - 5j, 0.0 + 0.02j, 7 - 7j], dtype=cdt, device=dev) + yft = torch.tensor([-1 + 5j, 3 - 3j, 5 - 5j], dtype=cdt, device=dev) + ytf = torch.tensor([-1 + 5j, 4 - 4j, 6 - 6j], dtype=cdt, device=dev) + ytt = torch.tensor([1 - 5j, 8 - 8j, 0.0 + 0.03j], dtype=cdt, device=dev) + sn_mva = 100.0 + + out = compute_flows(V, yff, yft, ytf, ytt, b_from, b_to, vn_kv, sn_mva) + + Vd = V.detach() + zero = torch.zeros(2, dtype=cdt, device=dev) + I_or = torch.stack([yff[0] * Vd[:, 0] + yft[0] * Vd[:, 1], + yff[1] * Vd[:, 2], + zero], dim=1) + I_ex = torch.stack([ytf[0] * Vd[:, 0] + ytt[0] * Vd[:, 1], + zero, + ytt[2] * Vd[:, 1]], dim=1) + V_or = torch.stack([Vd[:, 0], Vd[:, 2], zero], dim=1) + V_ex = torch.stack([Vd[:, 1], zero, Vd[:, 1]], dim=1) + base = sn_mva * 1e6 / (np.sqrt(3.0) * vn_kv[[0, 2, 1]] * 1e3) + S_or = V_or * I_or.conj() + S_ex = V_ex * I_ex.conj() + torch.testing.assert_close(out["i_or_a"], I_or.abs() * base) + torch.testing.assert_close(out["i_ex_a"], I_ex.abs() * base) + torch.testing.assert_close(out["p_or_mw"], S_or.real * sn_mva) + torch.testing.assert_close(out["q_or_mvar"], S_or.imag * sn_mva) + torch.testing.assert_close(out["p_ex_mw"], S_ex.real * sn_mva) + torch.testing.assert_close(out["q_ex_mvar"], S_ex.imag * sn_mva) + + # The -1 side contributes nothing, and nothing reaches the last bus + # (only branch 1's live end, bus 2, and bus 1 / bus 0 are read). + sum(v.sum() for v in out.values()).backward() + assert torch.isfinite(torch.view_as_real(V.grad)).all() + assert torch.all(V.grad[:, 3] == 0) diff --git a/tests/python/test_scenario_sweep_persistence.py b/tests/python/test_scenario_sweep_persistence.py index 730db7c..6c0dbf0 100644 --- a/tests/python/test_scenario_sweep_persistence.py +++ b/tests/python/test_scenario_sweep_persistence.py @@ -407,3 +407,117 @@ def test_set_branch_data_after_a_run_reaches_the_driver(self, ieee14_base_case, sw.compute_flows() np.testing.assert_allclose(sw.or_amps.to_numpy(), 2 * or1, rtol=solver_atol, atol=solver_atol) np.testing.assert_allclose(sw.ex_amps.to_numpy(), 2 * ex1, rtol=solver_atol, atol=solver_atol) + + @needs_bridge + def test_warm_run_outgrowing_the_cold_capacity_rebuilds_the_driver(self, solver_atol): + # The cold run sizes the chunk capacity for ITS active rows. A warm + # run with far more active rows must not be split into one-row chunks. + from gpusim2grid import ScenarioSweepGPU + grid, _, spur_line, _ = _solved_spur_grid(distributed_slack=False) + spur = int(spur_line) + scales = [1.0, 1.05, 0.95, 1.1, 0.9, 1.02] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(*_rows(grid, scales)) + + sw.set_topology([[spur]] * 5 + [[]]) # 1 active row of 6 + sw.compute(batch_size=6) + assert sw.solver.n_active == 1 and sw.solver.capacity == 1 + assert sw.driver_build_counter == 1 + + # mild growth (1 -> 2 active): one extra chunk, no new analysis + topo_mild = [[spur]] * 4 + [[], []] + sw.set_topology(topo_mild) + V_mild = _np(sw.compute(batch_size=6)) + assert sw.driver_build_counter == 1 and sw.source_build_counter == 2 + assert sw.solver.capacity == 1 + + # all 6 active: 6 chunks at the live capacity vs 1 fresh -> rebuild + topo_all = [[]] * 6 + sw.set_topology(topo_all) + V_all = _np(sw.compute(batch_size=6)) + assert sw.driver_build_counter == 2 + assert sw.solver.n_active == 6 and sw.solver.capacity == 6 + assert list(sw.get_disconnected()) == [0] * 6 + + ref_mild = _fresh(grid, scales, topology=topo_mild)[0] + ref_all = _fresh(grid, scales, topology=topo_all)[0] + np.testing.assert_allclose(V_mild[4:], ref_mild[4:], atol=solver_atol) + np.testing.assert_allclose(V_all, ref_all, atol=solver_atol) + + @needs_bridge + def test_turning_limit_violations_off_disarms_the_reused_driver(self, ieee14_base_case): + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL, + compute_limit_violations=True) + sw.set_injections_from_elements(*_rows(grid, [1.0, 1.1])) + sw.compute(batch_size=2) + t_on = sw.timings.t_violation_check + assert t_on.wall_ms > 0.0 or t_on.gpu_ms > 0.0 + + sw.compute_limit_violations = False + sw.compute(batch_size=2) + assert sw.driver_build_counter == 1 # same driver + t_off = sw.timings.t_violation_check + assert t_off.wall_ms == 0.0 and t_off.gpu_ms == 0.0 + with pytest.raises(RuntimeError, match="compute_limit_violations"): + sw.get_violations() + + def test_set_branch_data_refuses_ids_the_topology_still_trips(self, ieee14_base_case, + solver_atol): + from gpusim2grid import ScenarioSweepGPU + from gpusim2grid._ls2g_utils import extract_branch_data + grid = ieee14_base_case["grid"] + scales = [1.0, 1.1] + topo = [[3], []] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.set_injections_from_elements(*_rows(grid, scales)) + sw.set_topology(topo) + V1 = _np(sw.compute(batch_size=2)) + + args, _, _ = extract_branch_data(grid) + short = tuple(np.asarray(a)[:3] for a in args[:6]) + tuple(args[6:]) + with pytest.raises(RuntimeError, match="out of range"): + sw.set_branch_data(*short) + + # refused before anything was replaced: same topology, same answer + V2 = _np(sw.compute(batch_size=2)) + np.testing.assert_allclose(V2, V1, atol=solver_atol) + np.testing.assert_allclose(V2, _fresh(grid, scales, topology=topo)[0], + atol=solver_atol) + + @needs_bridge + def test_set_branch_data_with_more_branches(self, ieee14_base_case, solver_atol): + # One extra (zero-admittance) branch: the flow buffers are resized, + # the old per-branch limits are dropped (set_limits() is asked for + # again instead of reading past their end), the topology still holds. + from gpusim2grid import ScenarioSweepGPU + from gpusim2grid._ls2g_utils import extract_branch_data + grid = ieee14_base_case["grid"] + topo = [[3], []] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL, + compute_limit_violations=True) + sw.set_injections_from_elements(*_rows(grid, [1.0, 1.1])) + sw.set_topology(topo) + V1 = _np(sw.compute(batch_size=2)) + sw.compute_flows() + or1 = sw.or_amps.to_numpy().reshape(2, -1).copy() + n_bra = or1.shape[1] + + (b_from, b_to, yff, yft, ytf, ytt, vn_kv, sn_mva), _, _ = extract_branch_data(grid) + z = np.zeros(1, dtype=np.asarray(yff).dtype) + sw.set_branch_data(np.append(b_from, 0), np.append(b_to, 1), + np.append(yff, z), np.append(yft, z), + np.append(ytf, z), np.append(ytt, z), vn_kv, sn_mva) + with pytest.raises(RuntimeError, match="set_limits"): + sw.compute(batch_size=2) + + sw.compute_limit_violations = False + V2 = _np(sw.compute(batch_size=2)) + np.testing.assert_allclose(V2, V1, atol=solver_atol) + sw.compute_flows() + or2 = sw.or_amps.to_numpy().reshape(2, -1) + assert or2.shape == (2, n_bra + 1) + np.testing.assert_allclose(or2[:, :n_bra], or1, rtol=solver_atol, atol=solver_atol) + assert np.all(or2[:, n_bra] == 0.0) + assert or2[0, 3] == 0.0 # still tripped in row 0 From 0d2899291717c0f4cf84840969afc3ca3526dec6 Mon Sep 17 00:00:00 2001 From: Benjamin Donnot Date: Tue, 29 Sep 2026 18:39:29 +0000 Subject: [PATCH 7/7] fix adjoint, gen_v and NaN-residual issues found in a third review Each fix comes with a regression test. - BatchPfDriver::solve_JT_batch: refuse a forward that ran in several chunks. Every adjoint buffer holds one chunk; with a caller-supplied J snapshot (which passes the shape check) the rhs gather wrote n_active rows into it, out of bounds, and the gen_v kernel read past d_V_batch / d_Ybus_values_batch. - ScenarioSweepGPU: send gen_v to the session with the generators that set_contingency_gens() takes out of a row NaN'd there (re-sent when the mask changes). The conflict check already ignored them, but the session still applied their set-point, so an off generator could impose its voltage on a bus another connected generator keeps PV. - ScenarioSweepSession::run(): with keep_final_jacobian, also rebuild the driver when the live capacity needs more than one chunk and a fresh driver would not; the forward otherwise refused a batch_size that was already large enough, on every topology change. - ScenarioSweepSession::run(): count run_counter on entry and clear last_run_kept_jacobian_. A run() that threw after replacing the source or the driver left the counter unchanged, so a pending alias-mode BatchPowerFlow backward read the new buffers as its own. - acpf_nr.cu: the single-system ||F||_inf reductions (NR loop and the presolved check) used thrust::maximum from 0, which drops NaN: an all-NaN F read as 0 and was reported converged. They now propagate NaN like the batched compute_residuals_kernel. test_matching_alg's NaN case now also asserts converged is False; docstrings updated. Assisted-by: Claude Code Claude-Session: https://claude.ai/code/session_01FZwHuL9k42bdoU93GqaPQW Signed-off-by: Benjamin Donnot --- src/_cpp/acpf_nr.cu | 16 +++++++- src/_cpp/contingency/batch_pf_driver.cu | 12 ++++++ src/_cpp/python_bindings.cpp | 7 ++-- src/_cpp/scenario_sweep_session.cu | 16 +++++++- src/gpusim2grid/acpf_nr/gpu_facade.py | 6 +-- src/gpusim2grid/scenario_sweep/gpu_facade.py | 17 +++++++- tests/python/test_batch_power_flow.py | 39 ++++++++++++++++++ tests/python/test_gen_v.py | 41 +++++++++++++++++++ tests/python/test_matching_alg.py | 16 ++++---- .../python/test_scenario_sweep_persistence.py | 24 +++++++++++ 10 files changed, 174 insertions(+), 20 deletions(-) diff --git a/src/_cpp/acpf_nr.cu b/src/_cpp/acpf_nr.cu index 2610cc0..9cd907b 100644 --- a/src/_cpp/acpf_nr.cu +++ b/src/_cpp/acpf_nr.cu @@ -66,6 +66,7 @@ #include #include #include +#include // isnan / NAN (NanMaxFunctor) // --------------------------------------------------------------------------- // Error-checking macros — throw on failure so the AcPfNrState constructor @@ -109,6 +110,17 @@ struct AbsFunctor { cuda_real_type operator()(cuda_real_type x) const { return fabs(x); } }; +// NaN-propagating max for the ‖F‖∞ reductions below. thrust::maximum is +// `a < b ? b : a`, which drops a NaN `b`: an all-NaN F would reduce to the +// initial 0 and read as converged. Same sticky-NaN rule as the batched +// compute_residuals_kernel. +struct NanMaxFunctor { + __host__ __device__ + cuda_real_type operator()(cuda_real_type a, cuda_real_type b) const { + return (isnan(a) || isnan(b)) ? cuda_real_type(NAN) : (b > a ? b : a); + } +}; + // ============================================================================= // §2 CPU helper: build_J_structure // Derives the J CSR sparsity skeleton and the four dS scatter maps from the @@ -1181,7 +1193,7 @@ AcPfNrState::AcPfNrState( d_F.begin(), d_F.end(), AbsFunctor{}, cuda_real_type(0.), - thrust::maximum()); + NanMaxFunctor{}); timings.t_mismatch += timer.stop_ms(); // Guard against tol being tuned for lightsim2grid's FP64 CPU check @@ -1236,7 +1248,7 @@ AcPfNrState::AcPfNrState( d_F.begin(), d_F.end(), AbsFunctor{}, cuda_real_type(0.), - thrust::maximum()); + NanMaxFunctor{}); timings.t_mismatch += timer.stop_ms(); if (norm_F < static_cast(tol)) { diff --git a/src/_cpp/contingency/batch_pf_driver.cu b/src/_cpp/contingency/batch_pf_driver.cu index 23f6e13..1d154ef 100644 --- a/src/_cpp/contingency/batch_pf_driver.cu +++ b/src/_cpp/contingency/batch_pf_driver.cu @@ -1156,6 +1156,18 @@ void BatchPfDriver::solve_JT_batch( if (n_solves_ == 0) throw std::runtime_error( "[batch_pf] solve_JT_batch: call solve() (a forward run) first"); + // Every adjoint buffer (rhs, λ, the Jᵀ values, the gen_v scratch) holds + // ONE chunk: a forward split over several chunks has no single J to + // transpose, and gathering its n_active rows would write past them -- + // with the driver's own J (keep_final_jacobian refuses that forward) and + // with a caller-supplied snapshot alike. + if (n_active_ > batch_size_) + throw std::runtime_error( + "[batch_pf] solve_JT_batch: the last forward ran in several chunks (" + + std::to_string(n_active_) + " active rows, chunk capacity " + + std::to_string(batch_size_) + "); the adjoint needs one chunk -- raise " + "batch_size to at least the number of scenarios (or set " + "fixed_batch_capacity) and run() again"); if (!adjoint_) _prepare_adjoint(); BatchAdjoint& A = *adjoint_; diff --git a/src/_cpp/python_bindings.cpp b/src/_cpp/python_bindings.cpp index e156506..5cf1e80 100644 --- a/src/_cpp/python_bindings.cpp +++ b/src/_cpp/python_bindings.cpp @@ -664,10 +664,9 @@ PYBIND11_MODULE(_gpusim2grid, m) "matching_alg : MatchingAlg, CUDSS_CONFIG_MATCHING_ALG choice, same " "scope as reordering_alg. Default (NoMatching): cuDSS's own default " "(matching off). WARNING: MaxDiagProduct and Auto have been observed " - "to silently produce NaN voltages on real power-flow Jacobians while " - "AcPfTimings.converged still reports True (the ||F||_inf check does " - "not catch NaN) -- do not use them without independently checking " - "np.isnan(V) yourself. NoMatching/MaxDiagCount/MaxMinDiag/" + "to produce NaN voltages on real power-flow Jacobians (reported as " + "AcPfTimings.converged == False) -- do not rely on them. " + "NoMatching/MaxDiagCount/MaxMinDiag/" "MaxMinDiagAlt/MaxDiagSum have been verified to reproduce the " "reference solution.\n" "pivot_epsilon_alg: PivotEpsilonAlg, CUDSS_CONFIG_PIVOT_EPSILON_ALG " diff --git a/src/_cpp/scenario_sweep_session.cu b/src/_cpp/scenario_sweep_session.cu index db18c71..74c45a5 100644 --- a/src/_cpp/scenario_sweep_session.cu +++ b/src/_cpp/scenario_sweep_session.cu @@ -687,6 +687,13 @@ void ScenarioSweepSession::set_limits( // ============================================================================= void ScenarioSweepSession::run() { + // Counted on entry, not on success: a run() that throws part-way may + // already have swapped the batch source or overwritten the chunk + // buffers, so a pending alias-mode backward (run_counter guard) must no + // longer trust them -- nor the Jacobian of the previous forward. + ++run_counter_; + last_run_kept_jacobian_ = false; + if (!has_injections_) throw std::runtime_error( "ScenarioSweepSession: call set_injections() before run()"); @@ -861,7 +868,13 @@ void ScenarioSweepSession::run() if (!ctg.disconnected) ++n_active; const int chunks_live = (n_active + live_capacity - 1) / live_capacity; const int chunks_fresh = (n_active + batch_size_ - 1) / batch_size_; - if (!fixed_batch_capacity_ && chunks_live > 2 * chunks_fresh) { + // keep_final_jacobian needs ONE chunk: rebuild whenever a fresh driver + // would give it one and the live capacity would not (the forward + // would otherwise refuse a batch_size that is already large enough). + const bool too_many_chunks = + chunks_live > 2 * chunks_fresh + || (keep_final_jacobian_ && chunks_live > 1 && chunks_fresh <= 1); + if (!fixed_batch_capacity_ && too_many_chunks) { build_driver = true; reset_rows(); // the cold source below recomputes them } else { @@ -1007,7 +1020,6 @@ void ScenarioSweepSession::run() injections_dirty_ = topology_dirty_ = gen_v_dirty_ = gen_off_dirty_ = false; skip_dirty_ = false; last_run_kept_jacobian_ = keep_final_jacobian_; - ++run_counter_; solver_->cs.synchronize(); } diff --git a/src/gpusim2grid/acpf_nr/gpu_facade.py b/src/gpusim2grid/acpf_nr/gpu_facade.py index 0e615f6..7c143be 100644 --- a/src/gpusim2grid/acpf_nr/gpu_facade.py +++ b/src/gpusim2grid/acpf_nr/gpu_facade.py @@ -75,10 +75,8 @@ class AcPfGPU: ``"max_diag_sum"``, ``"max_diag_product"``, ``"auto"``. Construction-time only, same reason as ``reordering_alg``. WARNING: ``"max_diag_product"``/``"auto"`` have been observed to - silently produce NaN voltages on real power-flow Jacobians while - ``timings.converged`` still reports True (the residual check does - not catch NaN) — verify ``np.isnan(V).any()`` yourself if you use - them. ``"none"``/``"max_diag_count"``/``"max_min_diag"``/ + produce NaN voltages on real power-flow Jacobians (reported as + ``timings.converged == False``) — do not rely on them. ``"none"``/``"max_diag_count"``/``"max_min_diag"``/ ``"max_min_diag_alt"``/``"max_diag_sum"`` have been verified to reproduce the reference solution. pivot_epsilon_alg : str, default "default" diff --git a/src/gpusim2grid/scenario_sweep/gpu_facade.py b/src/gpusim2grid/scenario_sweep/gpu_facade.py index 9306ec7..f680d11 100644 --- a/src/gpusim2grid/scenario_sweep/gpu_facade.py +++ b/src/gpusim2grid/scenario_sweep/gpu_facade.py @@ -371,6 +371,8 @@ def set_contingency_gens(self, mask): self._gen_off = mask if self._pending_elements is not None: self._assemble_injections() + if self._gen_v is not None: + self._push_gen_v() # the off columns changed self._update_skipped_rows() @property @@ -431,10 +433,23 @@ def set_gen_v(self, gen_v): "set_gen_v() needs a lightsim2grid grid; explicit-array " "(tuple) mode has no generators to read.") gen_v = np.ascontiguousarray(gen_v, dtype=np.float64) - self._inner.set_gen_v(gen_v, self._elements.gen_bus) self._gen_v = gen_v + self._push_gen_v() self._update_skipped_rows() + def _push_gen_v(self): + """Hand set_gen_v()'s input to the session with the generators that + set_contingency_gens() takes out of a row NaN'd there (NaN = leave + untouched): a disconnected generator must not impose its set-point on + a bus another, still-connected generator keeps PV -- the conflict + check already ignores it, so the session must too. A row-count + mismatch is left to the C++ session, like _update_skipped_rows.""" + gen_v = self._gen_v + gen_off = self._gen_off + if gen_off is not None and gen_off.shape == gen_v.shape and gen_off.any(): + gen_v = np.where(gen_off, np.nan, gen_v) + self._inner.set_gen_v(gen_v, self._elements.gen_bus) + def _update_skipped_rows(self): """Re-derive the not-simulable rows from set_gen_v() (and the generator mask); a row-count mismatch is left to the C++ session.""" diff --git a/tests/python/test_batch_power_flow.py b/tests/python/test_batch_power_flow.py index 48fe2d5..3feb5a4 100644 --- a/tests/python/test_batch_power_flow.py +++ b/tests/python/test_batch_power_flow.py @@ -348,6 +348,45 @@ def test_solve_JT_batch_matches_spsolve(self): np.testing.assert_allclose(lam[r], expected, atol=1e-8, rtol=1e-8) assert np.all(lam[2] == 0.0) # dropped row + def test_solve_JT_batch_refuses_a_multi_chunk_forward(self, ieee14_base_case): + # The adjoint buffers hold one chunk: a snapshot J passes the shape + # check (it IS one chunk's worth), but gathering the n_active rows of + # a two-chunk forward into them would write out of bounds. + from gpusim2grid import ScenarioSweepGPU + grid = ieee14_base_case["grid"] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + n = 4 + load_p, load_q = (np.asarray(a) for a in grid.get_loads_res_full()[:2]) + gen_p = np.asarray(grid.get_gen_target_p()) + rep = lambda a: np.repeat(a[None, :], n, axis=0) # noqa: E731 + sw.set_injections_from_elements(rep(load_p), rep(load_q), rep(gen_p)) + sw.compute(batch_size=2) + sol = sw.solver + assert sol.capacity == 2 and sol.n_active == 4 + J = torch.from_dlpack(sol.j_values_dlpack()).clone() + rhs = torch.zeros(n, sol.dim_J, dtype=RDT, device="cuda") + with pytest.raises(RuntimeError, match="one chunk"): + sol.solve_JT_batch_dlpack(rhs.__dlpack__(), J.__dlpack__()) + + def test_failed_forward_invalidates_a_pending_alias_backward(self, ieee14_base_case): + # A forward that throws after the session already replaced its driver + # (here: keep_final_jacobian on a batch forced into two chunks) must + # still bump run_counter, or the pending backward would read the new + # driver's buffers as if they were its own. + grid = ieee14_base_case["grid"] + pf = _pf(grid) + n = 3 + load_p, load_q, gen_p = _base_inputs(pf, n, [1.0, 1.05, 0.95]) + lp = load_p.clone().requires_grad_(True) + V = pf(load_p=lp, load_q=load_q, gen_p=gen_p) + + pf.sweep.solver.batch_size = 1 # cold rebuild, 3 chunks + lp2 = load_p.clone().requires_grad_(True) + with pytest.raises(RuntimeError, match="keep_final_jacobian"): + pf(load_p=lp2, load_q=load_q, gen_p=gen_p) + with pytest.raises(RuntimeError, match="another forward"): + V.real.sum().backward() + # --------------------------------------------------------------------------- # Gradients vs finite differences diff --git a/tests/python/test_gen_v.py b/tests/python/test_gen_v.py index 64c9704..96f2dc3 100644 --- a/tests/python/test_gen_v.py +++ b/tests/python/test_gen_v.py @@ -312,6 +312,47 @@ def test_scenario_sweep_row_is_not_simulated(self, solver_atol): V3 = sw.solver.V_results.to_numpy().reshape(n, n_bus) np.testing.assert_allclose(np.abs(V3[1, b5]), 1.02, atol=solver_atol) + @pytest.mark.parametrize("mask_first", [False, True]) + def test_disconnected_generator_does_not_impose_its_setpoint(self, solver_atol, + mask_first): + # A generator taken out by set_contingency_gens must not write its + # gen_v onto a bus another connected generator keeps PV -- whichever + # of the two columns the kernel happens to write last, and whichever + # of set_gen_v / set_contingency_gens comes first. + from gpusim2grid import ScenarioSweepGPU + grid, (g_a, g_b), b5 = _two_gen_case14() + n_bus = grid.get_Ybus_solver().shape[0] + load_p, load_q = grid.get_loads_res_full()[:2] + gen_p = np.asarray(grid.get_gen_target_p()) + n = 3 + rep = lambda a: np.repeat(np.asarray(a)[None, :], n, axis=0) # noqa: E731 + n_gen = len(grid.get_generators()) + gen_v = np.full((n, n_gen), np.nan) + gen_v[:, [g_a, g_b]] = 1.02 + gen_v[1, g_a] = 1.05 # row 1: g_a off, g_b on at 1.02 + gen_v[2, g_b] = 1.05 # row 2: g_b off, g_a on at 1.02 + mask = np.zeros((n, n_gen), dtype=bool) + mask[1, g_a] = True + mask[2, g_b] = True + + sw = ScenarioSweepGPU(grid, nb_iter=10, tol_base=1e-10) + sw.set_injections_from_elements(rep(load_p), rep(load_q), rep(gen_p)) + if mask_first: + sw.set_contingency_gens(mask) + sw.set_gen_v(gen_v) + else: + sw.set_gen_v(gen_v) + sw.set_contingency_gens(mask) + sw.compute(batch_size=n) + assert sw.get_disconnected().tolist() == [0, 0, 0] + V = sw.solver.V_results.to_numpy().reshape(n, n_bus) + np.testing.assert_allclose(np.abs(V[:, b5]), [1.02] * n, atol=solver_atol) + + # the mask lifted: both generators are connected again and disagree + sw.set_contingency_gens(np.zeros((n, n_gen), dtype=bool)) + sw.compute(batch_size=n) + assert sw.get_disconnected().tolist() == [0, 1, 1] + def test_injection_sweep_raises(self): from gpusim2grid import InjectionSweepGPU grid, (g_a, g_b), _ = _two_gen_case14() diff --git a/tests/python/test_matching_alg.py b/tests/python/test_matching_alg.py index 8e550a4..4de8dbc 100644 --- a/tests/python/test_matching_alg.py +++ b/tests/python/test_matching_alg.py @@ -21,8 +21,8 @@ - Single-system (AcPfGPU): 'none', 'max_diag_count', 'max_min_diag', 'max_min_diag_alt', 'max_diag_sum' all reproduce the reference solution. 'max_diag_product' and 'auto' silently produce NaN voltages (all non-slack - buses) while ``timings.converged`` still reports True -- the ||F||_inf - residual check does not catch NaN (NaN comparisons are always False). This + buses). ``timings.converged`` now reports False for them (the ||F||_inf + reduction propagates NaN; it used to drop it and report True). This is a cuDSS/matching-scaling interaction with gpusim2grid's power-flow Jacobians, not a gpusim2grid plumbing bug -- see test_single_system_max_diag_product_and_auto_produce_nan below. Do not @@ -73,11 +73,12 @@ def test_converges_for_each_safe_alg(self, ieee14_base_case, solver_atol, alg): def test_single_system_max_diag_product_and_auto_produce_nan( self, ieee14_base_case, alg): """Documents a real (non-plumbing) finding: cuDSS's MaxDiagProduct/Auto - matching silently produces NaN voltages for gpusim2grid's power-flow - Jacobians, and the existing ||F||_inf convergence check does not catch - it (NaN comparisons are always False). This is not something to - silently work around here -- callers must be warned (see docstrings) - rather than have gpusim2grid quietly "fix" or hide it.""" + matching produces NaN voltages for gpusim2grid's power-flow + Jacobians. This is not something to silently work around here -- + callers must be warned (see docstrings) rather than have gpusim2grid + quietly "fix" or hide it -- but it must not be reported as converged + either: the ||F||_inf reduction propagates NaN (it used to drop it, + so an all-NaN F read as 0 and ``timings.converged`` was True).""" from gpusim2grid import AcPfGPU d = ieee14_base_case @@ -90,6 +91,7 @@ def test_single_system_max_diag_product_and_auto_produce_nan( f"matching_alg={alg!r} was expected to reproduce the known NaN " "issue; if this now passes, cuDSS behavior may have changed -- " "update _SINGLE_SYSTEM_SAFE_ALGS / the docstrings accordingly.") + assert not ac.timings.converged def test_enum_passthrough(self, ieee14_base_case, solver_atol): from gpusim2grid import AcPfGPU diff --git a/tests/python/test_scenario_sweep_persistence.py b/tests/python/test_scenario_sweep_persistence.py index 6c0dbf0..bc220f0 100644 --- a/tests/python/test_scenario_sweep_persistence.py +++ b/tests/python/test_scenario_sweep_persistence.py @@ -521,3 +521,27 @@ def test_set_branch_data_with_more_branches(self, ieee14_base_case, solver_atol) np.testing.assert_allclose(or2[:, :n_bra], or1, rtol=solver_atol, atol=solver_atol) assert np.all(or2[:, n_bra] == 0.0) assert or2[0, 3] == 0.0 # still tripped in row 0 + + @needs_bridge + def test_keep_final_jacobian_rebuilds_when_one_chunk_is_possible(self, solver_atol): + # keep_final_jacobian needs one chunk. The cold run's capacity (2: two + # of four rows islanded) would split a later all-active topology in two + # chunks, which is under the "twice as many chunks" rebuild threshold; + # the forward must still get the one chunk batch_size allows. + from gpusim2grid import ScenarioSweepGPU + grid, _, spur_line, _ = _solved_spur_grid(distributed_slack=False) + spur = int(spur_line) + scales = [1.0, 1.05, 0.95, 1.1] + sw = ScenarioSweepGPU(grid, nb_iter=NB_ITER, tol_base=TOL) + sw.solver.keep_final_jacobian = True + sw.set_injections_from_elements(*_rows(grid, scales)) + sw.set_topology([[spur], [spur], [], []]) + sw.compute(batch_size=4) + assert sw.solver.capacity == 2 + + topo = [[]] * 4 + sw.set_topology(topo) + V = _np(sw.compute(batch_size=4)) # used to raise + assert sw.driver_build_counter == 2 and sw.solver.capacity == 4 + np.testing.assert_allclose(V, _fresh(grid, scales, topology=topo)[0], + atol=solver_atol)