diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 2cedd83..d4c679a 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -8,6 +8,11 @@ updates: - "examples/**" - "tests/fixtures/**" + - package-ecosystem: "pip" + directory: "/projects/precipitation-anomaly-diagnostics" + schedule: + interval: "weekly" + - package-ecosystem: "github-actions" directory: "/" schedule: diff --git a/.github/workflows/precipitation-anomaly-diagnostics-ci.yml b/.github/workflows/precipitation-anomaly-diagnostics-ci.yml new file mode 100644 index 0000000..3936597 --- /dev/null +++ b/.github/workflows/precipitation-anomaly-diagnostics-ci.yml @@ -0,0 +1,53 @@ +name: precipitation-anomaly-diagnostics-ci +run-name: precipitation anomaly diagnostics ci / ${{ github.event_name }} / ${{ github.ref_name }} + +on: + workflow_dispatch: + push: + paths: + - ".github/workflows/precipitation-anomaly-diagnostics-ci.yml" + - "projects/precipitation-anomaly-diagnostics/**" + pull_request: + paths: + - ".github/workflows/precipitation-anomaly-diagnostics-ci.yml" + - "projects/precipitation-anomaly-diagnostics/**" + +permissions: {} + +env: + PRECIPITATION_DIAGNOSTICS_PYTHON_VERSION: "3.11" + +jobs: + test: + runs-on: ubuntu-latest + permissions: + contents: read + defaults: + run: + working-directory: projects/precipitation-anomaly-diagnostics + steps: + - name: Check out repository + uses: actions/checkout@v6 + + - name: Set up Python + uses: actions/setup-python@v6 + with: + python-version: ${{ env.PRECIPITATION_DIAGNOSTICS_PYTHON_VERSION }} + + - name: Upgrade pip + run: python -m pip install --upgrade pip + + - name: Install project + run: python -m pip install -e .[dev] + + - name: Run tests + run: python -m pytest + + - name: Compile modules and scripts + run: python -m compileall src scripts + + - name: CLI help smoke tests + run: | + python scripts/run_preprocessing.py --help + python scripts/run_eof_analysis.py --help + python scripts/run_composite_analysis.py --help diff --git a/README.md b/README.md index 99cd88a..671f389 100644 --- a/README.md +++ b/README.md @@ -13,9 +13,17 @@ can optionally record PyPI provenance and OpenSSF Scorecard evidence. For a fast reviewer overview, start with the [`sbom-diff-and-risk` reviewer brief](tools/sbom-diff-and-risk/docs/reviewer-brief.md). - -## Why This Repository Exists - + +## Additional Project + +[`projects/precipitation-anomaly-diagnostics`](projects/precipitation-anomaly-diagnostics/README.md) +is a public-safe climate diagnostics mini-lab. It demonstrates a reproducible +workflow for precipitation anomaly preprocessing, EOF analysis, representative +year selection, circulation composites, and reviewer-friendly scientific +interpretation. + +## Why This Repository Exists + Scientific and security-oriented engineering often needs small, inspectable tools that make evidence easier to review. This repository collects projects that emphasize: @@ -39,13 +47,31 @@ Deterministic SBOM/dependency diffing, JSON/Markdown/SARIF output, local policy checks, policy decision explainability, optional provenance and Scorecard evidence. -Useful entry points: - +Useful entry points: + - [`sbom-diff-and-risk` README](tools/sbom-diff-and-risk/README.md) - [Reviewer brief](tools/sbom-diff-and-risk/docs/reviewer-brief.md) - [Reviewer evidence pack](tools/sbom-diff-and-risk/docs/reviewer-evidence-pack.md) - [v0.9.0 release notes][release-notes-v090] - [Examples](tools/sbom-diff-and-risk/examples/) + +Project: +[`precipitation-anomaly-diagnostics`](projects/precipitation-anomaly-diagnostics/README.md) + +Status: +Public-safe mini-lab. + +What to review: +Sanitized climate-diagnostics workflow, small derived example artifacts, +methodology notes, data policy, and synthetic-data tests. + +Useful entry points: + +- [`precipitation-anomaly-diagnostics` README](projects/precipitation-anomaly-diagnostics/README.md) +- [Data policy](projects/precipitation-anomaly-diagnostics/docs/data-policy.md) +- [Methodology](projects/precipitation-anomaly-diagnostics/docs/methodology.md) +- [Inference framework](projects/precipitation-anomaly-diagnostics/docs/inference-framework.md) +- [Example report](projects/precipitation-anomaly-diagnostics/reports/example-report.md) ## Verification and Release Evidence diff --git a/projects/precipitation-anomaly-diagnostics/.gitignore b/projects/precipitation-anomaly-diagnostics/.gitignore new file mode 100644 index 0000000..530d46c --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/.gitignore @@ -0,0 +1,53 @@ +# Python +__pycache__/ +*.py[cod] +*.egg-info/ +.venv/ +venv/ +env/ +.pytest_cache/ +.mypy_cache/ +.ruff_cache/ + +# Local configs and outputs +configs/local*.yaml +data/ +raw/ +intermediate/ +outputs/ +tmp/ +scratch/ + +# Large climate data +*.nc +*.nc4 +*.grib +*.grb +*.hdf +*.h5 +*.zarr/ + +# Private / course artifacts +*.doc +*.docx +*.ppt +*.pptx +*.pdf +*.rar +*.zip +*.7z + +# Secrets and local environment +.env +.env.* +*.key +*.pem +*.p12 +*token* +*secret* + +# OS / editor +.DS_Store +Thumbs.db +.vscode/ +.idea/ diff --git a/projects/precipitation-anomaly-diagnostics/LICENSE b/projects/precipitation-anomaly-diagnostics/LICENSE new file mode 100644 index 0000000..3d2b178 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 stacknil + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/projects/precipitation-anomaly-diagnostics/README.md b/projects/precipitation-anomaly-diagnostics/README.md new file mode 100644 index 0000000..40d35dd --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/README.md @@ -0,0 +1,119 @@ +# Precipitation Anomaly Diagnostics Lab + +A compact scientific-computing workflow for July precipitation anomaly diagnostics over eastern China. + +This mini-lab demonstrates how to turn gridded precipitation and atmospheric circulation fields into reproducible climate diagnostics: + +- preprocess monthly precipitation and circulation fields; +- compute climatology, anomalies, and standardized anomaly indices; +- apply EOF analysis to identify dominant spatial modes; +- select representative years from standardized principal components; +- run circulation composite analysis for positive and negative phases; +- generate reviewable figures and lightweight Markdown reports. + +The repository is designed as a reproducible research mini-lab, not as an operational forecast system. + +## Repository Layout + +```text +. +├─ assets/sanitized-figures/ # derived, metadata-stripped demonstration figures +├─ configs/example.yaml # placeholder local data paths and analysis settings +├─ docs/ +│ ├─ data-policy.md +│ ├─ inference-framework.md +│ ├─ methodology.md +│ └─ reproducibility.md +├─ examples/ +│ ├─ regional_precipitation_summary_1961_2022.csv +│ └─ sample_metadata.json +├─ reports/example-report.md +├─ scripts/ +│ ├─ run_composite_analysis.py +│ ├─ run_eof_analysis.py +│ └─ run_preprocessing.py +└─ src/climate_diagnostics/ + ├─ composite.py + ├─ config.py + ├─ eof.py + ├─ plotting.py + └─ preprocess.py +``` + +## Installation + +Use Python 3.10+. + +```bash +python -m venv .venv +. .venv/bin/activate +pip install -e . +``` + +On Windows PowerShell: + +```powershell +python -m venv .venv +.\.venv\Scripts\Activate.ps1 +pip install -e . +``` + +Optional plotting dependencies such as `cartopy` may require platform-specific geospatial libraries. The core numerical workflow uses `numpy`, `pandas`, `xarray`, `scipy`, `matplotlib`, and `pyyaml`. + +## Expected Inputs + +This repository does not redistribute raw climate datasets. Users should obtain datasets from their original providers and configure local paths in `configs/example.yaml`. + +Typical inputs: + +- monthly precipitation fields with `time`, `lat`, and `lon` dimensions; +- atmospheric circulation fields such as `uwnd`, `vwnd`, `hgt`, and `omega`; +- a target month, region bounds, and reference climatology period. + +## Example Usage + +```bash +python scripts/run_preprocessing.py --config configs/example.yaml +python scripts/run_eof_analysis.py --config configs/example.yaml +python scripts/run_composite_analysis.py --config configs/example.yaml +``` + +The example CSV under `examples/` is a small derived summary table for demonstration only. It is not a raw dataset. + +## Demonstration Figures + +The `assets/sanitized-figures/` directory contains derived figures suitable for a public project page: + +- regional precipitation time series; +- decadal precipitation and category counts; +- EOF spatial regression modes; +- Monte Carlo variance screening; +- standardized EOF1 principal component; +- circulation climatology and phase-composite panels. + +## Diagnostic Inference + +The public version keeps a small amount of scientific interpretation while avoiding institutional or personal context. The key inference pattern is: + +```text +climatology -> anomaly field -> EOF modes -> screened signals -> representative years -> circulation composites -> mechanism hypothesis +``` + +The main methodological takeaway is that EOF modes should be treated as diagnostic coordinates, not as final explanations. Physical interpretation is added only after checking variance contribution, Monte Carlo screening, representative-year behavior, and vertically coherent circulation composites. + +See `docs/inference-framework.md` for the reusable reasoning framework. + +## Limitations + +- EOF signs are arbitrary. This project normalizes the first EOF mode so that positive PC values correspond to positive regional precipitation anomalies. +- Representative-year thresholds are configurable and should be treated as analysis choices, not universal physical constants. +- Composite diagnostics are descriptive and should be interpreted with physical context and uncertainty checks. +- The repository does not include raw climate data and cannot be fully reproduced until users provide compatible local datasets. + +## Data Policy + +Raw gridded climate datasets can be large and may have separate access policies. This repository does not redistribute CN05.1, NCEP/NCAR Reanalysis, or any other raw third-party climate dataset. See `docs/data-policy.md`. + +## Public-Safe Scope + +This public version is framed as a neutral scientific-computing project. It excludes course documents, institutional templates, personal identifiers, local-machine paths, and raw restricted datasets. diff --git a/projects/precipitation-anomaly-diagnostics/SANITIZATION_REPORT.md b/projects/precipitation-anomaly-diagnostics/SANITIZATION_REPORT.md new file mode 100644 index 0000000..36bb41e --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/SANITIZATION_REPORT.md @@ -0,0 +1,46 @@ +# Sanitization Report + +This working-tree report records how the public-safe export was prepared. + +## Files Inspected + +- root-level source materials; +- generated figures; +- generated reports; +- compressed data and analysis artifacts; +- public export tree. + +## Identifiers Removed or Excluded + +- personal names and student identifiers; +- institution, college, course, and classroom context; +- team-member and instructor references; +- local Windows and cloud-sync paths; +- course templates, slides, and word-processing reports; +- raw compressed data archives and large climate datasets. + +## Public Assets Kept + +- reusable Python source modules; +- neutral documentation; +- placeholder configuration; +- small derived CSV summary; +- derived demonstration figures with neutral filenames. + +## Raw Files Excluded + +- raw or packaged climate datasets; +- original course PDFs; +- original presentation decks; +- original word-processing reports; +- compressed working archives. + +## Remaining Assumptions + +- Included figures are derived scientific outputs and do not contain personal or institutional branding. +- The public repository should be created from this sanitized export directory, not from the original working folder. +- Git history should be clean because the export directory is separate from the original materials. + +## Recommended Review Path + +Publish or review this project from the sanitized subproject directory only. Do not publish the original working directory or raw source materials. diff --git a/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/circulation-climatology-panel.png b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/circulation-climatology-panel.png new file mode 100644 index 0000000..b9395c5 Binary files /dev/null and b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/circulation-climatology-panel.png differ diff --git a/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/decadal-precipitation-category-counts.png b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/decadal-precipitation-category-counts.png new file mode 100644 index 0000000..103c2a9 Binary files /dev/null and b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/decadal-precipitation-category-counts.png differ diff --git a/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof-modes-1-3-regression.png b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof-modes-1-3-regression.png new file mode 100644 index 0000000..31de8cf Binary files /dev/null and b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof-modes-1-3-regression.png differ diff --git a/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof-variance-monte-carlo.png b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof-variance-monte-carlo.png new file mode 100644 index 0000000..e561d33 Binary files /dev/null and b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof-variance-monte-carlo.png differ diff --git a/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof1-standardized-pc.png b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof1-standardized-pc.png new file mode 100644 index 0000000..fd2217b Binary files /dev/null and b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/eof1-standardized-pc.png differ diff --git a/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/negative-phase-circulation-composite.png b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/negative-phase-circulation-composite.png new file mode 100644 index 0000000..574aab8 Binary files /dev/null and b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/negative-phase-circulation-composite.png differ diff --git a/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/positive-phase-circulation-composite.png b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/positive-phase-circulation-composite.png new file mode 100644 index 0000000..80f2b2c Binary files /dev/null and b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/positive-phase-circulation-composite.png differ diff --git a/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/regional-july-precipitation-timeseries.png b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/regional-july-precipitation-timeseries.png new file mode 100644 index 0000000..f0db2f7 Binary files /dev/null and b/projects/precipitation-anomaly-diagnostics/assets/sanitized-figures/regional-july-precipitation-timeseries.png differ diff --git a/projects/precipitation-anomaly-diagnostics/configs/example.yaml b/projects/precipitation-anomaly-diagnostics/configs/example.yaml new file mode 100644 index 0000000..b646410 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/configs/example.yaml @@ -0,0 +1,32 @@ +project: + name: precipitation-anomaly-diagnostics + +paths: + precipitation_file: /path/to/monthly_precipitation.nc + circulation_dir: /path/to/reanalysis_fields/ + output_dir: outputs/ + +variables: + precipitation: pre + u_wind: uwnd + v_wind: vwnd + geopotential_height: hgt + omega: omega + +analysis: + month: 7 + climatology_period: [1961, 2022] + region: + lon_min: 105.0 + lon_max: 140.25 + lat_min: 14.75 + lat_max: 42.5 + eof: + n_modes: 10 + typical_year_threshold: 0.9 + monte_carlo_iterations: 1000 + monte_carlo_quantile: 0.95 + random_seed: 2026 + +notes: + raw_data_policy: Raw datasets must be obtained from original providers and are not redistributed by this repository. diff --git a/projects/precipitation-anomaly-diagnostics/docs/data-policy.md b/projects/precipitation-anomaly-diagnostics/docs/data-policy.md new file mode 100644 index 0000000..6065a85 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/docs/data-policy.md @@ -0,0 +1,38 @@ +# Data Policy + +This repository does not redistribute raw climate datasets. + +## Raw Data + +Users are expected to obtain gridded precipitation and reanalysis datasets from original providers and follow the corresponding licensing, citation, and access policies. + +Examples of compatible data types: + +- gridded monthly precipitation datasets; +- NCEP/NCAR-style reanalysis fields; +- wind, geopotential height, and omega variables. + +## What Is Included + +The repository includes only: + +- small derived CSV summaries; +- derived demonstration figures; +- reusable code; +- configuration templates; +- public-facing documentation. + +## What Is Excluded + +The repository intentionally excludes: + +- raw NetCDF or GRIB climate datasets; +- institutional course documents; +- presentation decks and word-processing reports; +- local-machine paths; +- personal identifiers; +- credentials or tokens. + +## Reproducing the Workflow + +To reproduce the full workflow, place locally obtained datasets outside the repository or under a gitignored `data/` directory, then point `configs/local.yaml` to those paths. diff --git a/projects/precipitation-anomaly-diagnostics/docs/inference-framework.md b/projects/precipitation-anomaly-diagnostics/docs/inference-framework.md new file mode 100644 index 0000000..3c60530 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/docs/inference-framework.md @@ -0,0 +1,125 @@ +# Inference Framework + +This project is not only a plotting workflow. It is a small example of how to turn gridded climate fields into a defensible diagnostic argument. + +The core methodological idea is: + +```text +spatial anomaly field +-> dominant modes +-> statistically screened signals +-> representative years +-> circulation composites +-> physically consistent mechanism hypothesis +``` + +The goal is not to claim deterministic causality from one dataset. The goal is to build a transparent chain of evidence that links precipitation anomalies to interpretable circulation patterns. + +## 1. Start From the Climate State + +Before anomaly diagnosis, establish the background state: + +- where the mean precipitation maxima are located; +- where interannual variability is strongest; +- whether the regional mean is dominated by a few extreme years; +- whether the target region has coherent or fragmented precipitation behavior. + +For July precipitation over eastern China, the demonstration results suggest a south-to-north gradient in mean precipitation and stronger variability over monsoon-affected southern and central-eastern subregions. This matters because high mean rainfall regions often also have stronger interannual variability, making them more sensitive to circulation shifts. + +Methodological takeaway: + +> Do not interpret anomalies before understanding the climatological background. An anomaly is meaningful only relative to the local seasonal climate state. + +## 2. Use EOF as Signal Compression, Not as the Answer + +EOF analysis is useful because it compresses a high-dimensional anomaly field into a small number of spatial modes and time coefficients. In this workflow: + +- EOF1 captures the dominant coherent precipitation anomaly pattern; +- EOF2 captures a secondary contrast pattern, such as north-south redistribution; +- EOF3 may capture more regional or pathway-dependent variability. + +The first mode in this example explains the largest share of variance and can be interpreted as a broad wet/dry phase over the core monsoon rainfall region. The second and third modes should be treated as redistribution patterns rather than weaker versions of EOF1. + +Methodological takeaway: + +> EOF modes should be read as diagnostic coordinates. They reveal structured variability, but their physical meaning must be checked against maps, time coefficients, and circulation fields. + +## 3. Screen Statistical Structure Before Explaining It + +Variance contribution alone is not enough. A mode with apparent structure can still be close to what a randomized field would produce. This project uses Monte Carlo screening by shuffling time independently at each grid point and comparing observed EOF variance against a random baseline. + +This step helps separate spatially coherent variability from chance alignment. + +Methodological takeaway: + +> A mode should not be over-interpreted unless it is both statistically distinguishable and physically interpretable. + +## 4. Convert Modes Into Representative Years + +The standardized principal component provides a way to select years that strongly express a mode. A threshold such as `0.9` standard deviations is not universal, but it makes the selection rule explicit and reproducible. + +Representative years are not simply the wettest or driest years by regional average. They are years whose spatial anomaly field projects strongly onto a particular EOF pattern. + +Methodological takeaway: + +> Representative years should be selected from the mode being studied, not only from a regional-mean index. This keeps the composite analysis aligned with the spatial pattern of interest. + +## 5. Use Composites as Physical Consistency Checks + +Composite analysis asks whether positive and negative EOF phases are associated with coherent atmospheric differences. + +For this demonstration, the physically interpretable pattern is a three-layer circulation chain: + +- upper troposphere: South Asian High configuration and high-level divergence; +- mid troposphere: western Pacific subtropical high position and strength; +- lower troposphere: East Asian summer monsoon and moisture transport; +- vertical motion: ascent or descent over the rainfall anomaly region. + +Positive EOF1 phase years are interpreted as wet-phase years when the circulation configuration favors moisture transport and ascent. Negative EOF1 phase years are interpreted as dry-phase years when the configuration weakens moisture supply and vertical lifting. + +Methodological takeaway: + +> Composite fields are most useful when multiple atmospheric layers tell a consistent story. A single map is weak evidence; a vertically coherent pattern is stronger diagnostic evidence. + +## 6. Separate Diagnosis From Causal Claims + +This workflow supports a mechanism hypothesis: + +```text +upper-level divergence ++ mid-level subtropical-high placement ++ low-level monsoon moisture transport ++ enhanced ascent +-> coherent positive precipitation anomalies +``` + +The opposite configuration is consistent with negative precipitation anomalies. + +However, this is still diagnostic inference. Stronger causal claims would require additional testing, such as: + +- moisture budget decomposition; +- wave activity or teleconnection diagnostics; +- sea-surface temperature forcing analysis; +- sensitivity experiments with numerical models; +- out-of-sample validation; +- comparison across multiple datasets. + +Methodological takeaway: + +> A good diagnostic project should state the difference between "consistent with" and "caused by." EOF and composites can motivate mechanisms, but they do not prove causality by themselves. + +## 7. A Reusable Mini-Lab Pattern + +This project can be reused as a general pattern for spatiotemporal anomaly diagnostics: + +1. Define the domain and season. +2. Compute climatology and anomalies. +3. Quantify regional mean behavior and variability hotspots. +4. Apply EOF or another dimensionality-reduction method. +5. Screen modes statistically. +6. Select representative years or events from standardized time coefficients. +7. Build composite anomaly fields. +8. Check whether the composites form a physically coherent mechanism. +9. Document assumptions, thresholds, and limits. + +This is the main public-facing value of the project: it is a compact example of how to move from raw gridded fields to interpretable, reviewable scientific reasoning. diff --git a/projects/precipitation-anomaly-diagnostics/docs/methodology.md b/projects/precipitation-anomaly-diagnostics/docs/methodology.md new file mode 100644 index 0000000..1618605 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/docs/methodology.md @@ -0,0 +1,56 @@ +# Methodology + +This mini-lab focuses on July precipitation anomaly diagnostics over eastern China using gridded climate data. + +## 1. Preprocessing + +1. Load monthly precipitation fields. +2. Select the target month, usually July. +3. Subset the target region. +4. Compute a reference climatology over a configurable period. +5. Compute anomalies as raw values minus climatology. + +For gridded spatial averages, latitude weights are computed with `cos(lat)`. + +## 2. EOF Analysis + +EOF analysis decomposes a space-time anomaly field into spatial modes and time coefficients. The workflow uses `sqrt(cos(lat))` area weighting before decomposition. + +Core outputs: + +- EOF spatial patterns; +- principal component time series; +- variance contribution by mode; +- cumulative explained variance. + +The first few modes are usually the most interpretable, but the exact number should be selected with variance contribution, statistical screening, and physical interpretation. + +## 3. Monte Carlo Screening + +Monte Carlo screening can be used to estimate whether observed EOF variance exceeds a random baseline. A common approach is: + +1. shuffle the time order independently at each grid point; +2. recompute EOF variance fractions; +3. repeat for many iterations; +4. compare observed variance against a selected quantile, such as the 95th percentile. + +This test evaluates whether spatially coherent variance is stronger than a randomized field with similar local variance. + +## 4. Representative Years + +Representative years are selected from a standardized principal component. In this project, a configurable threshold such as `0.9` standard deviations is used: + +- positive phase: standardized PC >= threshold; +- negative phase: standardized PC <= -threshold. + +EOF signs are arbitrary, so the project normalizes interpretation so that a positive EOF1 PC corresponds to positive regional precipitation anomalies. + +## 5. Circulation Composites + +Representative-year composites summarize atmospheric circulation differences between phases. Typical variables include: + +- `uwnd` and `vwnd` for wind fields; +- `hgt` for geopotential height; +- `omega` for vertical velocity. + +Composite maps are diagnostic summaries. They should be interpreted together with uncertainty checks and domain knowledge. diff --git a/projects/precipitation-anomaly-diagnostics/docs/reproducibility.md b/projects/precipitation-anomaly-diagnostics/docs/reproducibility.md new file mode 100644 index 0000000..69c5286 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/docs/reproducibility.md @@ -0,0 +1,52 @@ +# Reproducibility + +## Environment + +Create a Python environment and install the package: + +```bash +python -m venv .venv +. .venv/bin/activate +pip install -e . +``` + +Windows PowerShell: + +```powershell +python -m venv .venv +.\.venv\Scripts\Activate.ps1 +pip install -e . +``` + +## Configuration + +Copy `configs/example.yaml` to a local, gitignored file: + +```bash +cp configs/example.yaml configs/local.yaml +``` + +Edit local paths under `paths:` so they point to datasets stored outside the public repository or under a gitignored `data/` directory. + +## Pipeline + +```bash +python scripts/run_preprocessing.py --config configs/local.yaml +python scripts/run_eof_analysis.py --config configs/local.yaml +python scripts/run_composite_analysis.py --config configs/local.yaml --field /path/to/field.nc --variable hgt --years 1973,1975,1983 +``` + +## Public Release Checklist + +- [ ] No raw datasets included. +- [ ] No personal identifiers included. +- [ ] No institution or course identifiers included. +- [ ] No local absolute paths included. +- [ ] No course submission artifacts included. +- [ ] No secrets, API keys, credentials, or tokens included. +- [ ] No generated binary artifacts with unknown metadata included. +- [ ] All local configs are ignored by `.gitignore`. + +## Known Reproducibility Limits + +This public repository cannot fully reproduce the scientific results until users provide compatible local datasets. The included figures and CSV files are derived demonstration artifacts. diff --git a/projects/precipitation-anomaly-diagnostics/examples/README.md b/projects/precipitation-anomaly-diagnostics/examples/README.md new file mode 100644 index 0000000..b6a7afd --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/examples/README.md @@ -0,0 +1,8 @@ +# Examples + +This directory contains small derived artifacts for demonstration: + +- `regional_precipitation_summary_1961_2022.csv`: regional July precipitation summary; +- `sample_metadata.json`: lightweight metadata describing the example scope. + +These files are not raw climate datasets. diff --git a/projects/precipitation-anomaly-diagnostics/examples/regional_precipitation_summary_1961_2022.csv b/projects/precipitation-anomaly-diagnostics/examples/regional_precipitation_summary_1961_2022.csv new file mode 100644 index 0000000..4ff83db --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/examples/regional_precipitation_summary_1961_2022.csv @@ -0,0 +1,63 @@ +year,precip_mm,climatology_mm,anomaly_mm,anomaly_percent,category +1961,153.99,168.2,-14.22,-8.45,near_normal +1962,160.57,168.2,-7.64,-4.54,near_normal +1963,182.61,168.2,14.41,8.57,near_normal +1964,152.02,168.2,-16.19,-9.62,near_normal +1965,155.7,168.2,-12.5,-7.43,near_normal +1966,172.81,168.2,4.61,2.74,near_normal +1967,145.75,168.2,-22.45,-13.35,near_normal +1968,160.86,168.2,-7.35,-4.37,near_normal +1969,199.14,168.2,30.94,18.39,near_normal +1970,184.09,168.2,15.89,9.45,near_normal +1971,117.24,168.2,-50.96,-30.3,below_normal +1972,118.99,168.2,-49.21,-29.26,below_normal +1973,183.21,168.2,15.01,8.93,near_normal +1974,176.94,168.2,8.74,5.19,near_normal +1975,153.45,168.2,-14.75,-8.77,near_normal +1976,161.16,168.2,-7.04,-4.18,near_normal +1977,199.65,168.2,31.45,18.7,near_normal +1978,133.05,168.2,-35.15,-20.9,below_normal +1979,179.84,168.2,11.64,6.92,near_normal +1980,162.2,168.2,-6.0,-3.57,near_normal +1981,178.73,168.2,10.53,6.26,near_normal +1982,172.6,168.2,4.4,2.62,near_normal +1983,165.23,168.2,-2.97,-1.77,near_normal +1984,140.76,168.2,-27.44,-16.32,near_normal +1985,149.75,168.2,-18.45,-10.97,near_normal +1986,174.84,168.2,6.64,3.95,near_normal +1987,168.65,168.2,0.45,0.27,near_normal +1988,149.71,168.2,-18.49,-10.99,near_normal +1989,151.57,168.2,-16.63,-9.89,near_normal +1990,160.53,168.2,-7.67,-4.56,near_normal +1991,187.52,168.2,19.32,11.49,near_normal +1992,146.95,168.2,-21.25,-12.63,near_normal +1993,179.27,168.2,11.07,6.58,near_normal +1994,187.26,168.2,19.06,11.33,near_normal +1995,172.47,168.2,4.27,2.54,near_normal +1996,217.18,168.2,48.98,29.12,above_normal +1997,186.81,168.2,18.61,11.06,near_normal +1998,199.92,168.2,31.72,18.86,near_normal +1999,167.81,168.2,-0.39,-0.23,near_normal +2000,139.78,168.2,-28.42,-16.89,near_normal +2001,162.0,168.2,-6.2,-3.68,near_normal +2002,151.86,168.2,-16.34,-9.71,near_normal +2003,150.63,168.2,-17.57,-10.45,near_normal +2004,180.4,168.2,12.19,7.25,near_normal +2005,148.14,168.2,-20.06,-11.92,near_normal +2006,174.92,168.2,6.72,4.0,near_normal +2007,170.52,168.2,2.32,1.38,near_normal +2008,178.5,168.2,10.3,6.12,near_normal +2009,161.81,168.2,-6.39,-3.8,near_normal +2010,177.68,168.2,9.48,5.64,near_normal +2011,132.88,168.2,-35.32,-21.0,below_normal +2012,182.75,168.2,14.55,8.65,near_normal +2013,190.59,168.2,22.39,13.31,near_normal +2014,166.47,168.2,-1.73,-1.03,near_normal +2015,156.09,168.2,-12.11,-7.2,near_normal +2016,194.58,168.2,26.38,15.68,near_normal +2017,170.38,168.2,2.18,1.29,near_normal +2018,182.03,168.2,13.83,8.22,near_normal +2019,179.43,168.2,11.23,6.68,near_normal +2020,209.32,168.2,41.12,24.45,above_normal +2021,198.78,168.2,30.58,18.18,near_normal +2022,158.05,168.2,-10.15,-6.03,near_normal diff --git a/projects/precipitation-anomaly-diagnostics/examples/sample_metadata.json b/projects/precipitation-anomaly-diagnostics/examples/sample_metadata.json new file mode 100644 index 0000000..ad582d3 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/examples/sample_metadata.json @@ -0,0 +1,9 @@ +{ + "project": "precipitation-anomaly-diagnostics", + "domain": "scientific computing / climate diagnostics", + "region": "eastern China, east of 105E and south of 42.5N", + "period": "1961-2022", + "month": "July", + "raw_data_redistributed": false, + "notes": "Example CSV and figures are derived demonstration artifacts; raw datasets must be obtained from original providers." +} \ No newline at end of file diff --git a/projects/precipitation-anomaly-diagnostics/pyproject.toml b/projects/precipitation-anomaly-diagnostics/pyproject.toml new file mode 100644 index 0000000..d984f10 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/pyproject.toml @@ -0,0 +1,27 @@ +[build-system] +requires = ["setuptools>=68", "wheel"] +build-backend = "setuptools.build_meta" + +[project] +name = "precipitation-anomaly-diagnostics" +version = "0.1.0" +description = "A public-safe scientific-computing mini-lab for precipitation anomaly diagnostics." +readme = "README.md" +requires-python = ">=3.10" +license = { text = "MIT" } +authors = [{ name = "stacknil" }] +dependencies = [ + "numpy>=1.24", + "pandas>=2.0", + "xarray>=2023.1", + "scipy>=1.10", + "matplotlib>=3.7", + "pyyaml>=6.0" +] + +[project.optional-dependencies] +plot = ["cartopy>=0.22"] +dev = ["pytest>=7.0", "ruff>=0.4"] + +[tool.setuptools.packages.find] +where = ["src"] diff --git a/projects/precipitation-anomaly-diagnostics/reports/example-report.md b/projects/precipitation-anomaly-diagnostics/reports/example-report.md new file mode 100644 index 0000000..eede68f --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/reports/example-report.md @@ -0,0 +1,32 @@ +# Example Report Summary + +This example summarizes the public-safe interpretation of the diagnostic workflow. + +## Scope + +The workflow analyzes July precipitation anomalies over eastern China from a gridded monthly precipitation dataset. It computes climatology, anomaly percentages, EOF modes, representative years, and circulation composites. + +## Main Diagnostic Outputs + +- A regional July precipitation time series. +- EOF spatial modes and principal component time series. +- Monte Carlo screening for EOF variance. +- Representative positive and negative phase years. +- Circulation composites for phase interpretation. + +## Interpretation Notes + +The first EOF mode describes the dominant coherent precipitation anomaly pattern. Positive and negative phase years are selected from standardized principal components. Circulation composites provide a diagnostic view of atmospheric differences between phases, but should not be treated as causal proof without additional physical and statistical checks. + +## Methodological Takeaways + +The analysis is organized as an evidence chain rather than a single plot-based conclusion: + +1. Establish the July precipitation climatology and variability background. +2. Use EOF analysis to compress the anomaly field into interpretable spatial coordinates. +3. Screen EOF variance against a randomized Monte Carlo baseline. +4. Select representative years from standardized principal components. +5. Compare positive and negative phase circulation composites. +6. Treat the resulting circulation pattern as a mechanism hypothesis, not as standalone causal proof. + +The reusable lesson is that spatiotemporal climate diagnostics should combine statistical structure with physical consistency checks. In this example, a wet-phase interpretation is stronger when upper-level divergence, mid-level subtropical-high placement, low-level monsoon moisture transport, and vertical motion anomalies point in the same direction. diff --git a/projects/precipitation-anomaly-diagnostics/scripts/run_composite_analysis.py b/projects/precipitation-anomaly-diagnostics/scripts/run_composite_analysis.py new file mode 100644 index 0000000..f221719 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/scripts/run_composite_analysis.py @@ -0,0 +1,31 @@ +from __future__ import annotations + +import argparse + +import xarray as xr + +from climate_diagnostics.composite import composite_anomaly +from climate_diagnostics.config import ensure_output_dir, load_config + + +def main() -> None: + parser = argparse.ArgumentParser(description="Run a simple composite anomaly analysis.") + parser.add_argument("--config", required=True, help="Path to a YAML configuration file.") + parser.add_argument("--field", required=True, help="Input field NetCDF path.") + parser.add_argument("--variable", required=True, help="Variable name to composite.") + parser.add_argument("--years", required=True, help="Comma-separated representative years.") + parser.add_argument("--name", default="composite_anomaly", help="Output file stem.") + args = parser.parse_args() + + config = load_config(args.config) + output_dir = ensure_output_dir(config) + dataset = xr.open_dataset(args.field) + field = dataset[args.variable] + years = [int(value) for value in args.years.split(",") if value.strip()] + climatology = field.mean("time", skipna=True) + result = composite_anomaly(field, years, climatology) + result.to_dataset(name=args.variable).to_netcdf(output_dir / f"{args.name}.nc") + + +if __name__ == "__main__": + main() diff --git a/projects/precipitation-anomaly-diagnostics/scripts/run_eof_analysis.py b/projects/precipitation-anomaly-diagnostics/scripts/run_eof_analysis.py new file mode 100644 index 0000000..c3ddecf --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/scripts/run_eof_analysis.py @@ -0,0 +1,41 @@ +from __future__ import annotations + +import argparse + +import pandas as pd +import xarray as xr + +from climate_diagnostics.config import ensure_output_dir, load_config +from climate_diagnostics.eof import compute_eof, representative_years, standardized_pc + + +def main() -> None: + parser = argparse.ArgumentParser(description="Run EOF analysis on a precipitation anomaly dataset.") + parser.add_argument("--config", required=True, help="Path to a YAML configuration file.") + parser.add_argument("--input", default=None, help="Optional anomaly NetCDF path.") + args = parser.parse_args() + + config = load_config(args.config) + output_dir = ensure_output_dir(config) + eof_config = config["analysis"]["eof"] + anomaly_path = args.input or output_dir / "precipitation_anomaly.nc" + dataset = xr.open_dataset(anomaly_path) + field = next(iter(dataset.data_vars.values())) + + result = compute_eof(field, n_modes=int(eof_config["n_modes"])) + result.patterns.to_netcdf(output_dir / "eof_patterns.nc") + result.pcs.to_netcdf(output_dir / "eof_pcs.nc") + + variance = result.variance_fraction.to_series().reset_index() + variance["variance_percent"] = variance["variance_fraction"] * 100.0 + variance.to_csv(output_dir / "eof_variance.csv", index=False) + + pc1 = standardized_pc(result.pcs.sel(mode=1)) + years = representative_years(pc1, threshold=float(eof_config["typical_year_threshold"])) + pd.DataFrame( + [{"phase": phase, "years": ",".join(map(str, values))} for phase, values in years.items()] + ).to_csv(output_dir / "representative_years.csv", index=False) + + +if __name__ == "__main__": + main() diff --git a/projects/precipitation-anomaly-diagnostics/scripts/run_preprocessing.py b/projects/precipitation-anomaly-diagnostics/scripts/run_preprocessing.py new file mode 100644 index 0000000..63677f0 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/scripts/run_preprocessing.py @@ -0,0 +1,21 @@ +from __future__ import annotations + +import argparse + +from climate_diagnostics.config import ensure_output_dir, load_config +from climate_diagnostics.preprocess import build_precipitation_anomaly_dataset + + +def main() -> None: + parser = argparse.ArgumentParser(description="Build a regional precipitation anomaly dataset.") + parser.add_argument("--config", required=True, help="Path to a YAML configuration file.") + args = parser.parse_args() + + config = load_config(args.config) + output_dir = ensure_output_dir(config) + dataset = build_precipitation_anomaly_dataset(config) + dataset.to_netcdf(output_dir / "precipitation_anomaly.nc") + + +if __name__ == "__main__": + main() diff --git a/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/__init__.py b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/__init__.py new file mode 100644 index 0000000..faa58bd --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/__init__.py @@ -0,0 +1,9 @@ +"""Utilities for precipitation anomaly and circulation diagnostics.""" + +__all__ = [ + "config", + "preprocess", + "eof", + "composite", + "plotting", +] diff --git a/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/composite.py b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/composite.py new file mode 100644 index 0000000..18cc8a4 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/composite.py @@ -0,0 +1,17 @@ +from __future__ import annotations + +from collections.abc import Iterable + +import xarray as xr + + +def composite_by_years(field: xr.DataArray, years: Iterable[int]) -> xr.DataArray: + """Average a field over selected calendar years.""" + years_set = set(int(year) for year in years) + selected = field.sel(time=field["time.year"].isin(years_set)) + return selected.mean("time", skipna=True) + + +def composite_anomaly(field: xr.DataArray, years: Iterable[int], climatology: xr.DataArray) -> xr.DataArray: + """Composite anomaly relative to a provided climatology.""" + return composite_by_years(field, years) - climatology diff --git a/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/config.py b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/config.py new file mode 100644 index 0000000..b0b9c11 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/config.py @@ -0,0 +1,22 @@ +from __future__ import annotations + +from pathlib import Path +from typing import Any + +import yaml + + +def load_config(path: str | Path) -> dict[str, Any]: + """Load a YAML configuration file.""" + with Path(path).open("r", encoding="utf-8") as handle: + data = yaml.safe_load(handle) + if not isinstance(data, dict): + raise ValueError("Configuration must be a YAML mapping.") + return data + + +def ensure_output_dir(config: dict[str, Any]) -> Path: + """Create and return the configured output directory.""" + output_dir = Path(config["paths"]["output_dir"]) + output_dir.mkdir(parents=True, exist_ok=True) + return output_dir diff --git a/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/eof.py b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/eof.py new file mode 100644 index 0000000..2120460 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/eof.py @@ -0,0 +1,110 @@ +from __future__ import annotations + +from dataclasses import dataclass + +import numpy as np +import xarray as xr + + +@dataclass +class EOFResult: + patterns: xr.DataArray + pcs: xr.DataArray + variance_fraction: xr.DataArray + + +def latitude_weights(lat: xr.DataArray) -> xr.DataArray: + """Return sqrt(cos(lat)) EOF area weights.""" + return xr.DataArray(np.sqrt(np.cos(np.deg2rad(lat))), coords={"lat": lat}, dims="lat") + + +def _stack_valid_grid(field: xr.DataArray) -> tuple[xr.DataArray, xr.DataArray]: + stacked = field.stack(space=("lat", "lon")) + valid = stacked.notnull().all("time") + return stacked.where(valid, drop=True), valid + + +def compute_eof(field: xr.DataArray, n_modes: int = 10) -> EOFResult: + """Compute EOF modes with SVD. + + The input must have dimensions `time`, `lat`, and `lon`. + """ + if n_modes < 1: + raise ValueError("n_modes must be at least 1.") + weights = latitude_weights(field.lat) + weighted = field * weights + stacked, valid = _stack_valid_grid(weighted) + if stacked.sizes["space"] == 0: + raise ValueError("EOF analysis requires at least one grid cell without missing values.") + matrix = stacked.transpose("time", "space").values + matrix = matrix - np.nanmean(matrix, axis=0, keepdims=True) + mode_count = min(n_modes, *matrix.shape) + + u, singular_values, vt = np.linalg.svd(matrix, full_matrices=False) + pcs = u[:, :mode_count] * singular_values[:mode_count] + eigenvalues = singular_values**2 / max(matrix.shape[0] - 1, 1) + total_variance = eigenvalues.sum() + if not np.isfinite(total_variance) or total_variance <= 0: + raise ValueError("EOF analysis requires nonzero finite variance.") + variance_fraction = eigenvalues[:mode_count] / total_variance + + modes = np.arange(1, mode_count + 1) + pattern_values = np.full((mode_count, valid.sizes["space"]), np.nan) + pattern_values[:, valid.values] = vt[:mode_count, :] + patterns = xr.DataArray( + pattern_values, + dims=("mode", "space"), + coords={"mode": modes, "space": valid.space}, + ).unstack("space") + + pcs_da = xr.DataArray( + pcs, + dims=("time", "mode"), + coords={"time": field.time, "mode": modes}, + name="pc", + ) + vf_da = xr.DataArray( + variance_fraction, + dims=("mode",), + coords={"mode": modes}, + name="variance_fraction", + ) + return EOFResult(patterns=patterns, pcs=pcs_da, variance_fraction=vf_da) + + +def standardized_pc(pc: xr.DataArray) -> xr.DataArray: + """Return a standardized principal component time series.""" + return (pc - pc.mean("time")) / pc.std("time") + + +def representative_years(pc: xr.DataArray, threshold: float = 0.9) -> dict[str, list[int]]: + """Select positive and negative representative years from a standardized PC.""" + standardized = standardized_pc(pc) + years = standardized["time.year"].values + values = standardized.values + return { + "positive": [int(year) for year, value in zip(years, values) if value >= threshold], + "negative": [int(year) for year, value in zip(years, values) if value <= -threshold], + } + + +def monte_carlo_variance_threshold( + field: xr.DataArray, + n_modes: int = 10, + iterations: int = 1000, + quantile: float = 0.95, + seed: int | None = None, +) -> np.ndarray: + """Estimate EOF variance thresholds by shuffling time at each grid point.""" + rng = np.random.default_rng(seed) + stacked, _ = _stack_valid_grid(field) + matrix = stacked.transpose("time", "space").values + samples = np.empty((iterations, n_modes), dtype=float) + for i in range(iterations): + shuffled = matrix.copy() + for j in range(shuffled.shape[1]): + rng.shuffle(shuffled[:, j]) + _, singular_values, _ = np.linalg.svd(shuffled - shuffled.mean(axis=0, keepdims=True), full_matrices=False) + eigenvalues = singular_values**2 / max(shuffled.shape[0] - 1, 1) + samples[i, :] = eigenvalues[:n_modes] / eigenvalues.sum() + return np.quantile(samples, quantile, axis=0) diff --git a/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/plotting.py b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/plotting.py new file mode 100644 index 0000000..65ecbdf --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/plotting.py @@ -0,0 +1,24 @@ +from __future__ import annotations + +from pathlib import Path + +import matplotlib.pyplot as plt +import pandas as pd + + +def plot_regional_series(summary_csv: str | Path, output: str | Path) -> None: + """Plot regional precipitation and anomaly percentage from a summary CSV.""" + data = pd.read_csv(summary_csv) + fig, axes = plt.subplots(2, 1, figsize=(11, 7), sharex=True) + axes[0].plot(data["year"], data["precip_mm"], marker="o", lw=1.4) + axes[0].axhline(data["climatology_mm"].iloc[0], ls="--", color="0.4") + axes[0].set_ylabel("Precipitation (mm)") + axes[0].set_title("Regional July precipitation") + colors = ["#c63d3d" if value >= 0 else "#2f78b7" for value in data["anomaly_percent"]] + axes[1].bar(data["year"], data["anomaly_percent"], color=colors) + axes[1].axhline(0, color="0.2", lw=1) + axes[1].set_ylabel("Anomaly (%)") + axes[1].set_xlabel("Year") + fig.tight_layout() + fig.savefig(output, dpi=160, bbox_inches="tight", metadata={}) + plt.close(fig) diff --git a/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/preprocess.py b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/preprocess.py new file mode 100644 index 0000000..bb00034 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/src/climate_diagnostics/preprocess.py @@ -0,0 +1,62 @@ +from __future__ import annotations + +from pathlib import Path +from typing import Any + +import numpy as np +import xarray as xr + + +def _coordinate_slice(coord: xr.DataArray, lower: float, upper: float) -> slice: + if coord.size == 0: + raise ValueError("Cannot subset an empty coordinate.") + if float(coord[0]) <= float(coord[-1]): + return slice(lower, upper) + return slice(upper, lower) + + +def subset_region(field: xr.DataArray, region: dict[str, float]) -> xr.DataArray: + """Subset a gridded field to lon/lat bounds.""" + return field.sel( + lon=_coordinate_slice(field.lon, region["lon_min"], region["lon_max"]), + lat=_coordinate_slice(field.lat, region["lat_min"], region["lat_max"]), + ) + + +def select_month(field: xr.DataArray, month: int) -> xr.DataArray: + """Select one calendar month from a monthly time series.""" + return field.sel(time=field["time.month"] == month) + + +def climatology(field: xr.DataArray, start_year: int, end_year: int) -> xr.DataArray: + """Compute climatology over an inclusive year range.""" + selected = field.sel(time=slice(f"{start_year}-01-01", f"{end_year}-12-31")) + return selected.mean("time", skipna=True) + + +def anomaly(field: xr.DataArray, reference: xr.DataArray) -> xr.DataArray: + """Return anomalies relative to a reference climatology.""" + return field - reference + + +def area_weighted_mean(field: xr.DataArray) -> xr.DataArray: + """Compute a latitude-weighted spatial mean.""" + weights = xr.DataArray(np.cos(np.deg2rad(field.lat)), coords={"lat": field.lat}, dims="lat") + return field.weighted(weights).mean(("lat", "lon"), skipna=True) + + +def build_precipitation_anomaly_dataset(config: dict[str, Any]) -> xr.Dataset: + """Open precipitation data and build a regional monthly anomaly dataset.""" + path = Path(config["paths"]["precipitation_file"]) + var_name = config["variables"]["precipitation"] + month = int(config["analysis"]["month"]) + period = config["analysis"]["climatology_period"] + region = config["analysis"]["region"] + + dataset = xr.open_dataset(path) + precip = subset_region(dataset[var_name], region) + precip_month = select_month(precip, month) + reference = climatology(precip_month, int(period[0]), int(period[1])) + anomalies = anomaly(precip_month, reference) + anomalies.name = "precipitation_anomaly" + return anomalies.to_dataset() diff --git a/projects/precipitation-anomaly-diagnostics/tests/test_eof.py b/projects/precipitation-anomaly-diagnostics/tests/test_eof.py new file mode 100644 index 0000000..b6ef447 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/tests/test_eof.py @@ -0,0 +1,55 @@ +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest +import xarray as xr + +from climate_diagnostics.eof import compute_eof, representative_years, standardized_pc + + +def _sample_field() -> xr.DataArray: + time_signal = np.array([-2.0, -1.0, 1.0, 2.0]) + spatial_pattern = np.array([[1.0, 0.6], [-0.2, -0.8]]) + values = np.stack([value * spatial_pattern for value in time_signal]) + return xr.DataArray( + values, + dims=("time", "lat", "lon"), + coords={ + "time": pd.date_range("2001-07-01", periods=4, freq="YS"), + "lat": [25.0, 35.0], + "lon": [115.0, 125.0], + }, + ) + + +def test_compute_eof_limits_modes_to_available_rank() -> None: + result = compute_eof(_sample_field(), n_modes=10) + + assert result.patterns.sizes["mode"] == 4 + assert result.pcs.sizes == {"time": 4, "mode": 4} + assert result.variance_fraction.sizes == {"mode": 4} + np.testing.assert_allclose(float(result.variance_fraction.sum()), 1.0) + + +def test_compute_eof_rejects_empty_or_constant_fields() -> None: + all_nan = xr.full_like(_sample_field(), np.nan) + with pytest.raises(ValueError, match="at least one grid cell"): + compute_eof(all_nan) + + constant = xr.full_like(_sample_field(), 1.0) + with pytest.raises(ValueError, match="nonzero finite variance"): + compute_eof(constant) + + +def test_representative_years_from_standardized_pc() -> None: + pc = xr.DataArray( + [-2.0, -0.5, 0.25, 1.5], + dims=("time",), + coords={"time": pd.to_datetime(["2001-07-01", "2002-07-01", "2003-07-01", "2004-07-01"])}, + ) + + standardized = standardized_pc(pc) + result = representative_years(standardized, threshold=0.9) + + assert result == {"positive": [2004], "negative": [2001]} diff --git a/projects/precipitation-anomaly-diagnostics/tests/test_preprocess.py b/projects/precipitation-anomaly-diagnostics/tests/test_preprocess.py new file mode 100644 index 0000000..490fba2 --- /dev/null +++ b/projects/precipitation-anomaly-diagnostics/tests/test_preprocess.py @@ -0,0 +1,91 @@ +from __future__ import annotations + +import numpy as np +import pandas as pd +import xarray as xr + +from climate_diagnostics.preprocess import area_weighted_mean, build_precipitation_anomaly_dataset, subset_region + + +def test_subset_region_handles_descending_latitude() -> None: + field = xr.DataArray( + np.arange(3 * 4 * 5).reshape(3, 4, 5), + dims=("time", "lat", "lon"), + coords={ + "time": pd.date_range("2001-07-01", periods=3, freq="YS"), + "lat": [45.0, 35.0, 25.0, 15.0], + "lon": [100.0, 110.0, 120.0, 130.0, 140.0], + }, + ) + + result = subset_region( + field, + { + "lon_min": 105.0, + "lon_max": 135.0, + "lat_min": 20.0, + "lat_max": 40.0, + }, + ) + + assert result.lat.values.tolist() == [35.0, 25.0] + assert result.lon.values.tolist() == [110.0, 120.0, 130.0] + + +def test_area_weighted_mean_preserves_time_dimension() -> None: + field = xr.DataArray( + np.ones((2, 2, 2)), + dims=("time", "lat", "lon"), + coords={"time": [0, 1], "lat": [20.0, 30.0], "lon": [110.0, 120.0]}, + ) + + result = area_weighted_mean(field) + + assert result.dims == ("time",) + np.testing.assert_allclose(result.values, [1.0, 1.0]) + + +def test_build_precipitation_anomaly_dataset(tmp_path) -> None: + path = tmp_path / "precipitation.nc" + data = xr.Dataset( + { + "pre": xr.DataArray( + np.array( + [ + [[10.0, 20.0]], + [[30.0, 40.0]], + ] + ), + dims=("time", "lat", "lon"), + coords={ + "time": pd.to_datetime(["2001-07-01", "2002-07-01"]), + "lat": [30.0], + "lon": [120.0, 121.0], + }, + ) + } + ) + data.to_netcdf(path) + + result = build_precipitation_anomaly_dataset( + { + "paths": {"precipitation_file": str(path)}, + "variables": {"precipitation": "pre"}, + "analysis": { + "month": 7, + "climatology_period": [2001, 2002], + "region": { + "lon_min": 119.0, + "lon_max": 122.0, + "lat_min": 29.0, + "lat_max": 31.0, + }, + }, + } + ) + + assert set(result.data_vars) == {"precipitation_anomaly"} + np.testing.assert_allclose( + result["precipitation_anomaly"].values, + [[[-10.0, -10.0]], [[10.0, 10.0]]], + )