diff --git a/.github/workflows/backup-verification.yml b/.github/workflows/backup-verification.yml new file mode 100644 index 0000000..d839c62 --- /dev/null +++ b/.github/workflows/backup-verification.yml @@ -0,0 +1,54 @@ +name: Backup Verification + +# #1066: Weekly off-cluster restore test (the in-cluster CronJob runs daily). +# Running it from a second environment proves backups are restorable even if +# the production cluster is gone. + +on: + schedule: + - cron: "0 6 * * 1" + workflow_dispatch: + +permissions: + contents: read + id-token: write + +jobs: + verify: + runs-on: ubuntu-latest + timeout-minutes: 180 + environment: backup-verification + steps: + - uses: actions/checkout@v4 + + - uses: aws-actions/configure-aws-credentials@v4 + with: + role-to-assume: ${{ secrets.BACKUP_VERIFY_ROLE_ARN }} + aws-region: ${{ vars.BACKUP_REGION }} + + - name: Install tools + run: sudo apt-get update && sudo apt-get install -y zstd jq + + - name: Restore and verify latest backup + env: + BACKUP_BUCKET: ${{ secrets.BACKUP_BUCKET }} + PUSHGATEWAY_URL: ${{ secrets.PUSHGATEWAY_URL }} + ALERTMANAGER_URL: ${{ secrets.ALERTMANAGER_URL }} + run: scripts/ops/backup-verify.sh + + - name: Upload verification report + if: always() + uses: actions/upload-artifact@v4 + with: + name: backup-verification-${{ github.run_id }} + path: backup-verification.json + retention-days: 90 + + - name: Open issue on failure + if: failure() + env: + GH_TOKEN: ${{ github.token }} + run: | + gh issue create --title "Backup verification failed ($(date -u +%F))" \ + --label incident,backups \ + --body "Run: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}. See docs/backup-strategy.md §5." diff --git a/.github/workflows/cost-report.yml b/.github/workflows/cost-report.yml new file mode 100644 index 0000000..fcced4a --- /dev/null +++ b/.github/workflows/cost-report.yml @@ -0,0 +1,58 @@ +name: Cost Report + +# #1067: Weekly cost report generated from the savings ledger written by +# scripts/ops/cost-optimize.sh, published as a workflow summary + artifact. + +on: + schedule: + - cron: "0 7 * * 1" + workflow_dispatch: + inputs: + days: + description: Reporting window in days + default: "30" + +permissions: + contents: read + id-token: write + +jobs: + report: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: aws-actions/configure-aws-credentials@v4 + with: + role-to-assume: ${{ secrets.COST_REPORT_ROLE_ARN }} + aws-region: ${{ vars.BACKUP_REGION }} + + - name: Fetch savings ledger + run: aws s3 cp "${{ secrets.COST_LEDGER_URI }}" cost-ledger.jsonl + + - name: Build report + env: + COST_LEDGER: cost-ledger.jsonl + REPORT_DAYS: ${{ inputs.days || '30' }} + REPORT_OUT: cost-report.md + run: scripts/ops/cost-optimize.sh report + + - name: Append cloud spend (AWS Cost Explorer) + run: | + start=$(date -u -d "-${{ inputs.days || '30' }} days" +%F); end=$(date -u +%F) + aws ce get-cost-and-usage --time-period "Start=$start,End=$end" \ + --granularity MONTHLY --metrics UnblendedCost \ + --filter '{"Tags":{"Key":"project","Values":["atomicip"]}}' \ + --group-by Type=DIMENSION,Key=SERVICE \ + | jq -r '"\n## Cloud spend by service\n\n| Service | USD |\n|---|---:|", + (.ResultsByTime[].Groups[] | "| \(.Keys[0]) | \(.Metrics.UnblendedCost.Amount | tonumber * 100 | round / 100) |")' \ + >> cost-report.md + + - name: Publish summary + run: cat cost-report.md >> "$GITHUB_STEP_SUMMARY" + + - uses: actions/upload-artifact@v4 + with: + name: cost-report-${{ github.run_id }} + path: cost-report.md + retention-days: 365 diff --git a/.github/workflows/gitops-promote.yml b/.github/workflows/gitops-promote.yml new file mode 100644 index 0000000..55d8da7 --- /dev/null +++ b/.github/workflows/gitops-promote.yml @@ -0,0 +1,104 @@ +name: GitOps Promote + +# #1069: Pull-request based deployments. +# - push to main -> build image, commit new tag to the staging overlay +# (Argo CD auto-syncs staging) +# - workflow_dispatch -> open a PR bumping the production overlay to a +# given tag; merging that PR deploys production. +# Rollback is `git revert` of the promotion commit (scripts/gitops-rollback.sh). + +on: + push: + branches: [main] + paths: + - "api-server/**" + workflow_dispatch: + inputs: + tag: + description: Image tag (git SHA) already running in staging to promote to production + required: true + type: string + +permissions: + contents: write + packages: write + pull-requests: write + +concurrency: + group: gitops-promote + cancel-in-progress: false + +env: + IMAGE: ghcr.io/atomicip/api-server + +jobs: + build-and-stage: + if: github.event_name == 'push' + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - uses: docker/login-action@v3 + with: + registry: ghcr.io + username: ${{ github.actor }} + password: ${{ secrets.GITHUB_TOKEN }} + + - uses: docker/build-push-action@v6 + with: + context: api-server + push: true + tags: ${{ env.IMAGE }}:${{ github.sha }} + + - name: Install kustomize + run: | + curl -sSL "https://github.com/kubernetes-sigs/kustomize/releases/download/kustomize%2Fv5.4.3/kustomize_v5.4.3_linux_amd64.tar.gz" \ + | tar -xz -C /usr/local/bin kustomize + + - name: Bump staging overlay + shell: bash + run: | + set -euo pipefail + cd deploy/k8s/overlays/staging + kustomize edit set image "${IMAGE}=${IMAGE}:${GITHUB_SHA}" + cd - + git config user.name "atomicip-gitops-bot" + git config user.email "gitops-bot@users.noreply.github.com" + git add deploy/k8s/overlays/staging/kustomization.yaml + git commit -m "deploy(staging): api-server ${GITHUB_SHA::12}" + git push origin HEAD:main + + promote-production: + if: github.event_name == 'workflow_dispatch' + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + + - name: Verify image exists + run: docker manifest inspect "${IMAGE}:${{ inputs.tag }}" > /dev/null + + - name: Install kustomize + run: | + curl -sSL "https://github.com/kubernetes-sigs/kustomize/releases/download/kustomize%2Fv5.4.3/kustomize_v5.4.3_linux_amd64.tar.gz" \ + | tar -xz -C /usr/local/bin kustomize + + - name: Bump production overlay + working-directory: deploy/k8s/overlays/production + run: kustomize edit set image "${IMAGE}=${IMAGE}:${{ inputs.tag }}" + + - name: Open promotion pull request + uses: peter-evans/create-pull-request@v6 + with: + branch: gitops/promote-production-${{ inputs.tag }} + commit-message: "deploy(production): api-server ${{ inputs.tag }}" + title: "deploy(production): api-server ${{ inputs.tag }}" + labels: deployment, production + body: | + Promotes `${{ env.IMAGE }}:${{ inputs.tag }}` from staging to production. + + Merging this PR deploys it: Argo CD syncs `deploy/k8s/overlays/production` + automatically. To roll back, revert the merge commit + (`scripts/gitops-rollback.sh production`). + + - [ ] Tag verified healthy in staging + - [ ] No open SEV-1/SEV-2 incidents diff --git a/.github/workflows/gitops-validate.yml b/.github/workflows/gitops-validate.yml new file mode 100644 index 0000000..8861ed8 --- /dev/null +++ b/.github/workflows/gitops-validate.yml @@ -0,0 +1,65 @@ +name: GitOps Manifest Validation + +# #1069: Every pull request touching deployment manifests is rendered and +# schema-validated, and the rendered diff against main is posted for review. +# Merging the PR *is* the deployment: Argo CD syncs main to the cluster. + +on: + pull_request: + paths: + - "deploy/**" + +permissions: + contents: read + pull-requests: write + +jobs: + validate: + runs-on: ubuntu-latest + strategy: + matrix: + overlay: [staging, production] + steps: + - uses: actions/checkout@v4 + with: + fetch-depth: 0 + + - name: Install kustomize and kubeconform + shell: bash + run: | + set -euo pipefail + curl -sSL "https://github.com/kubernetes-sigs/kustomize/releases/download/kustomize%2Fv5.4.3/kustomize_v5.4.3_linux_amd64.tar.gz" \ + | tar -xz -C /usr/local/bin kustomize + curl -sSL "https://github.com/yannh/kubeconform/releases/download/v0.6.7/kubeconform-linux-amd64.tar.gz" \ + | tar -xz -C /usr/local/bin kubeconform + + - name: Render overlay + run: kustomize build --load-restrictor LoadRestrictionsNone "deploy/k8s/overlays/${{ matrix.overlay }}" > rendered.yaml + + - name: Validate schemas + run: kubeconform -strict -summary -ignore-missing-schemas rendered.yaml + + - name: Diff against main + id: diff + shell: bash + run: | + set -euo pipefail + git worktree add /tmp/base origin/${{ github.base_ref }} + if [ -d "/tmp/base/deploy/k8s/overlays/${{ matrix.overlay }}" ]; then + kustomize build --load-restrictor LoadRestrictionsNone "/tmp/base/deploy/k8s/overlays/${{ matrix.overlay }}" > base.yaml + else + : > base.yaml + fi + diff -u base.yaml rendered.yaml > manifest.diff || true + { + echo "### Rendered diff: \`${{ matrix.overlay }}\`" + echo '```diff' + head -c 60000 manifest.diff + echo '```' + } > comment.md + + - name: Comment diff on PR + uses: marocchino/sticky-pull-request-comment@v2 + with: + header: gitops-diff-${{ matrix.overlay }} + path: comment.md diff --git a/CHANGELOG.md b/CHANGELOG.md index 2bdda8c..d04a000 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -71,6 +71,12 @@ Each entry below references an issue number from our GitHub repository. When rev - **#781** — Arbitrator committee mechanism with M-of-N signatures and time-locked ruling enforcement - **#906** — Treasury address validation: guard against hardcoded placeholder addresses +### Operations & Reliability +- **#1066** — Automated backup verification: daily/weekly restore tests, integrity checks, RTO measurement, failure alerts (`scripts/ops/backup-verify.sh`, `docs/backup-strategy.md`) +- **#1067** — Cost optimization automation: api-server HPA, orphaned data cleanup, compression/tiering, savings ledger and reports (`scripts/ops/cost-optimize.sh`, `docs/cost-optimization.md`) +- **#1068** — Incident management: PagerDuty/Opsgenie integration, incidents from alerts, acknowledgment, postmortems (`api-server/src/incidents.rs`, `docs/incident-response.md`) +- **#1069** — GitOps deployment pipeline: Argo CD + Kustomize, PR-based promotion, rollback via `git revert` (`deploy/`, `docs/gitops.md`) + ## Contributing When adding new features or fixes, update this file with: diff --git a/api-server/src/commitment_monitoring.rs b/api-server/src/commitment_monitoring.rs index fcc2ace..07184a9 100644 --- a/api-server/src/commitment_monitoring.rs +++ b/api-server/src/commitment_monitoring.rs @@ -371,6 +371,8 @@ pub fn spawn_background_evaluator() { COMMITMENT_MONITOR.evaluate(now); alerting::ALERT_MANAGER.tick(now); for n in alerting::ALERT_MANAGER.drain_notifications() { + // #1068: critical notifications open (or re-trigger) an incident. + crate::incidents::INCIDENTS.open_from_notification(&n, now); tracing::warn!( alert = %n.alert_name, severity = ?n.severity, diff --git a/api-server/src/incidents.rs b/api-server/src/incidents.rs new file mode 100644 index 0000000..43f1dbe --- /dev/null +++ b/api-server/src/incidents.rs @@ -0,0 +1,755 @@ +//! #1068: Incident management integration. +//! +//! Turns alerts into tracked incidents and mirrors them to an external paging +//! provider (PagerDuty or Opsgenie). Alerts reach this module from two places: +//! +//! * the in-process pipeline in [`crate::alerting`] — critical notifications +//! drained by `commitment_monitoring::spawn_background_evaluator`; +//! * Prometheus Alertmanager, via the webhook receiver +//! `POST /v1/admin/incidents/alertmanager` (see `monitoring/alertmanager`). +//! +//! Each incident keeps a timeline (trigger, re-trigger, acknowledge, resolve, +//! notes) and, once resolved, a postmortem. Acknowledging or resolving an +//! incident here is propagated to the provider and to the originating alert so +//! escalation stops. +//! +//! Configuration (environment): +//! * `INCIDENT_PROVIDER` – `pagerduty`, `opsgenie` or `none` (default). +//! * `PAGERDUTY_ROUTING_KEY` – Events API v2 integration key. +//! * `OPSGENIE_API_KEY` / `OPSGENIE_API_URL` – GenieKey and API base +//! (default `https://api.opsgenie.com`, use `https://api.eu.opsgenie.com` for EU). +//! +//! Procedures: docs/incident-response.md. Postmortem template: +//! docs/postmortem-template.md. + +use std::collections::HashMap; +use std::sync::Mutex; + +use axum::extract::Path; +use axum::http::StatusCode; +use axum::Json; +use metrics::{counter, histogram}; +use once_cell::sync::Lazy; +use serde::{Deserialize, Serialize}; +use serde_json::json; + +use crate::alerting::{self, Notification, Severity}; + +// ── Model ──────────────────────────────────────────────────────────────────── + +/// Incident severity. SEV1 is the most severe. +#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum IncidentSeverity { + Sev1, + Sev2, + Sev3, +} + +impl IncidentSeverity { + /// Maps alert severity (plus escalation level) to incident severity. + pub fn from_alert(severity: Severity, escalation_level: usize) -> Self { + match (severity, escalation_level) { + (Severity::Critical, l) if l >= 2 => IncidentSeverity::Sev1, + (Severity::Critical, _) => IncidentSeverity::Sev2, + _ => IncidentSeverity::Sev3, + } + } + + fn from_label(label: Option<&str>) -> Self { + match label { + Some("critical") => IncidentSeverity::Sev2, + Some("sev1") | Some("page") => IncidentSeverity::Sev1, + _ => IncidentSeverity::Sev3, + } + } + + fn pagerduty(&self) -> &'static str { + match self { + IncidentSeverity::Sev1 | IncidentSeverity::Sev2 => "critical", + IncidentSeverity::Sev3 => "warning", + } + } + + fn opsgenie(&self) -> &'static str { + match self { + IncidentSeverity::Sev1 => "P1", + IncidentSeverity::Sev2 => "P2", + IncidentSeverity::Sev3 => "P3", + } + } + + /// SEV1/SEV2 incidents require a postmortem (docs/incident-response.md §6). + pub fn requires_postmortem(&self) -> bool { + *self <= IncidentSeverity::Sev2 + } + + fn as_str(&self) -> &'static str { + match self { + IncidentSeverity::Sev1 => "sev1", + IncidentSeverity::Sev2 => "sev2", + IncidentSeverity::Sev3 => "sev3", + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "lowercase")] +pub enum IncidentStatus { + Triggered, + Acknowledged, + Resolved, +} + +#[derive(Debug, Clone, Serialize)] +pub struct TimelineEntry { + pub at: u64, + pub kind: String, + pub actor: String, + pub message: String, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct ActionItem { + pub description: String, + pub owner: String, + /// Issue / ticket URL tracking the follow-up. + #[serde(default)] + pub tracking_url: Option, +} + +#[derive(Debug, Clone, Serialize, Deserialize)] +pub struct Postmortem { + pub author: String, + pub summary: String, + pub impact: String, + pub root_cause: String, + #[serde(default)] + pub contributing_factors: Vec, + #[serde(default)] + pub what_went_well: Vec, + #[serde(default)] + pub what_went_wrong: Vec, + #[serde(default)] + pub action_items: Vec, + #[serde(default)] + pub document_url: Option, + #[serde(default)] + pub submitted_at: u64, +} + +#[derive(Debug, Clone, Serialize)] +pub struct Incident { + pub id: String, + /// Stable key used to dedupe re-triggers and to address the incident at + /// the provider (PagerDuty `dedup_key` / Opsgenie `alias`). + pub dedup_key: String, + pub title: String, + pub severity: IncidentSeverity, + pub status: IncidentStatus, + pub source: String, + /// Fingerprint of the originating in-process alert, if any. + pub alert_fingerprint: Option, + /// Whether this incident is mirrored to the external provider. Incidents + /// from Alertmanager are not: Alertmanager pages the provider itself, and + /// mirroring them would page twice. + pub paged: bool, + pub trigger_count: u64, + pub created_at: u64, + pub acknowledged_at: Option, + pub acknowledged_by: Option, + pub resolved_at: Option, + pub resolved_by: Option, + pub timeline: Vec, + pub postmortem: Option, +} + +impl Incident { + fn push(&mut self, at: u64, kind: &str, actor: &str, message: impl Into) { + self.timeline.push(TimelineEntry { + at, + kind: kind.into(), + actor: actor.into(), + message: message.into(), + }); + } + + pub fn postmortem_due(&self) -> bool { + self.status == IncidentStatus::Resolved + && self.severity.requires_postmortem() + && self.postmortem.is_none() + } +} + +#[derive(Debug, Clone)] +pub struct NewIncident { + pub dedup_key: String, + pub title: String, + pub severity: IncidentSeverity, + pub source: String, + pub alert_fingerprint: Option, + pub paged: bool, + pub details: serde_json::Value, +} + +#[derive(Debug, PartialEq, Eq)] +pub enum IncidentError { + NotFound, + InvalidState(&'static str), +} + +// ── Provider integration ───────────────────────────────────────────────────── + +/// External paging provider. +#[derive(Debug, Clone)] +pub enum Provider { + None, + PagerDuty { routing_key: String }, + Opsgenie { api_key: String, base_url: String }, +} + +/// Lifecycle event to mirror to the provider. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum ProviderAction { + Trigger, + Acknowledge, + Resolve, +} + +impl Provider { + pub fn from_env() -> Self { + match std::env::var("INCIDENT_PROVIDER").unwrap_or_default().to_lowercase().as_str() { + "pagerduty" => match std::env::var("PAGERDUTY_ROUTING_KEY") { + Ok(routing_key) if !routing_key.is_empty() => Provider::PagerDuty { routing_key }, + _ => { + tracing::warn!("INCIDENT_PROVIDER=pagerduty but PAGERDUTY_ROUTING_KEY unset; paging disabled"); + Provider::None + } + }, + "opsgenie" => match std::env::var("OPSGENIE_API_KEY") { + Ok(api_key) if !api_key.is_empty() => Provider::Opsgenie { + api_key, + base_url: std::env::var("OPSGENIE_API_URL") + .unwrap_or_else(|_| "https://api.opsgenie.com".into()), + }, + _ => { + tracing::warn!("INCIDENT_PROVIDER=opsgenie but OPSGENIE_API_KEY unset; paging disabled"); + Provider::None + } + }, + _ => Provider::None, + } + } + + fn name(&self) -> &'static str { + match self { + Provider::None => "none", + Provider::PagerDuty { .. } => "pagerduty", + Provider::Opsgenie { .. } => "opsgenie", + } + } + + /// Build the HTTP request for `action`. Returns `None` when there is + /// nothing to send (no provider configured). + fn request( + &self, + client: &reqwest::Client, + action: ProviderAction, + incident: &Incident, + details: &serde_json::Value, + actor: &str, + ) -> Option { + match self { + Provider::None => None, + Provider::PagerDuty { routing_key } => { + let event_action = match action { + ProviderAction::Trigger => "trigger", + ProviderAction::Acknowledge => "acknowledge", + ProviderAction::Resolve => "resolve", + }; + let mut body = json!({ + "routing_key": routing_key, + "event_action": event_action, + "dedup_key": incident.dedup_key, + }); + if action == ProviderAction::Trigger { + body["payload"] = json!({ + "summary": incident.title, + "source": incident.source, + "severity": incident.severity.pagerduty(), + "component": "atomicip-api", + "group": "atomicip", + "custom_details": details, + }); + body["links"] = json!([{ + "href": "https://github.com/AtomicIP/AtomicIP-/blob/main/docs/incident-response.md", + "text": "Incident response procedures", + }]); + } + Some(client.post("https://events.pagerduty.com/v2/enqueue").json(&body)) + } + Provider::Opsgenie { api_key, base_url } => { + let alias = urlencode(&incident.dedup_key); + let req = match action { + ProviderAction::Trigger => client + .post(format!("{base_url}/v2/alerts")) + .json(&json!({ + "message": truncate(&incident.title, 130), + "alias": incident.dedup_key, + "source": incident.source, + "priority": incident.severity.opsgenie(), + "tags": ["atomicip", incident.severity.as_str()], + "details": flatten_details(details), + })), + ProviderAction::Acknowledge => client + .post(format!("{base_url}/v2/alerts/{alias}/acknowledge?identifierType=alias")) + .json(&json!({ "user": actor, "source": "atomicip-api" })), + ProviderAction::Resolve => client + .post(format!("{base_url}/v2/alerts/{alias}/close?identifierType=alias")) + .json(&json!({ "user": actor, "source": "atomicip-api" })), + }; + Some(req.header("Authorization", format!("GenieKey {api_key}"))) + } + } + } +} + +fn truncate(s: &str, max: usize) -> String { + s.chars().take(max).collect() +} + +fn urlencode(s: &str) -> String { + s.bytes() + .map(|b| match b { + b'A'..=b'Z' | b'a'..=b'z' | b'0'..=b'9' | b'-' | b'_' | b'.' | b'~' => (b as char).to_string(), + _ => format!("%{b:02X}"), + }) + .collect() +} + +/// Opsgenie `details` must be a flat string map. +fn flatten_details(details: &serde_json::Value) -> serde_json::Value { + match details.as_object() { + Some(map) => serde_json::Value::Object( + map.iter() + .map(|(k, v)| { + let s = v.as_str().map(str::to_owned).unwrap_or_else(|| v.to_string()); + (k.clone(), serde_json::Value::String(s)) + }) + .collect(), + ), + None => json!({}), + } +} + +// ── Manager ────────────────────────────────────────────────────────────────── + +pub struct IncidentManager { + provider: Provider, + client: reqwest::Client, + incidents: Mutex>, + /// dedup_key -> id of the currently open incident with that key. + open_by_key: Mutex>, +} + +impl IncidentManager { + pub fn new(provider: Provider) -> Self { + Self { + provider, + client: reqwest::Client::builder() + .timeout(std::time::Duration::from_secs(10)) + .build() + .unwrap_or_default(), + incidents: Mutex::new(HashMap::new()), + open_by_key: Mutex::new(HashMap::new()), + } + } + + pub fn provider_name(&self) -> &'static str { + self.provider.name() + } + + /// Open a new incident, or record a re-trigger on the open incident with + /// the same dedup key. Returns the incident id and whether it was new. + pub fn trigger(&self, new: NewIncident, now: u64) -> (String, bool) { + let mut open = self.open_by_key.lock().unwrap(); + let mut incidents = self.incidents.lock().unwrap(); + + if let Some(id) = open.get(&new.dedup_key) { + if let Some(inc) = incidents.get_mut(id) { + inc.trigger_count += 1; + // Severity only ever increases while the incident is open. + if new.severity < inc.severity { + inc.severity = new.severity; + let msg = format!("severity raised to {}", new.severity.as_str()); + inc.push(now, "severity_change", "system", msg); + } + inc.push(now, "retrigger", &new.source, new.title.clone()); + counter!("incidents_retriggered_total").increment(1); + return (id.clone(), false); + } + } + + let id = format!("INC-{}", uuid::Uuid::new_v4().simple().to_string()[..10].to_uppercase()); + let mut incident = Incident { + id: id.clone(), + dedup_key: new.dedup_key.clone(), + title: new.title, + severity: new.severity, + status: IncidentStatus::Triggered, + source: new.source.clone(), + alert_fingerprint: new.alert_fingerprint, + paged: new.paged, + trigger_count: 1, + created_at: now, + acknowledged_at: None, + acknowledged_by: None, + resolved_at: None, + resolved_by: None, + timeline: Vec::new(), + postmortem: None, + }; + incident.push(now, "triggered", &new.source, incident.title.clone()); + counter!("incidents_created_total", "severity" => incident.severity.as_str()).increment(1); + + self.notify(ProviderAction::Trigger, &incident, new.details, "atomicip-api"); + open.insert(new.dedup_key, id.clone()); + incidents.insert(id.clone(), incident); + (id, true) + } + + /// Create (or re-trigger) an incident from an in-process alert notification. + /// Only critical notifications page; everything else stays in the alert + /// pipeline. + pub fn open_from_notification(&self, n: &Notification, now: u64) -> Option { + if n.severity != Severity::Critical { + return None; + } + let (id, _) = self.trigger( + NewIncident { + dedup_key: format!("alert-{:016x}", n.fingerprint), + title: format!("{}: {}", n.alert_name, n.summary), + severity: IncidentSeverity::from_alert(n.severity, n.escalation_level), + source: "atomicip-alerting".into(), + alert_fingerprint: Some(n.fingerprint), + paged: true, + details: json!({ + "alert": n.alert_name, + "occurrences": n.occurrences, + "correlated_alerts": n.correlated_count, + "escalation_level": n.escalation_level, + "channel": n.channel, + }), + }, + now, + ); + Some(id) + } + + pub fn acknowledge(&self, id: &str, actor: &str, now: u64) -> Result { + let mut incidents = self.incidents.lock().unwrap(); + let inc = incidents.get_mut(id).ok_or(IncidentError::NotFound)?; + match inc.status { + IncidentStatus::Resolved => return Err(IncidentError::InvalidState("incident already resolved")), + IncidentStatus::Acknowledged => return Ok(inc.clone()), + IncidentStatus::Triggered => {} + } + inc.status = IncidentStatus::Acknowledged; + inc.acknowledged_at = Some(now); + inc.acknowledged_by = Some(actor.to_owned()); + inc.push(now, "acknowledged", actor, "incident acknowledged"); + histogram!("incident_time_to_acknowledge_seconds", "severity" => inc.severity.as_str()) + .record(now.saturating_sub(inc.created_at) as f64); + if let Some(fp) = inc.alert_fingerprint { + // Stops escalation in the in-process alert pipeline. + alerting::ALERT_MANAGER.acknowledge(fp); + } + self.notify(ProviderAction::Acknowledge, inc, json!({}), actor); + Ok(inc.clone()) + } + + pub fn resolve(&self, id: &str, actor: &str, note: Option, now: u64) -> Result { + // Lock order must match `trigger`: open_by_key before incidents. + let mut open = self.open_by_key.lock().unwrap(); + let mut incidents = self.incidents.lock().unwrap(); + let inc = incidents.get_mut(id).ok_or(IncidentError::NotFound)?; + if inc.status == IncidentStatus::Resolved { + return Ok(inc.clone()); + } + if inc.acknowledged_at.is_none() { + inc.acknowledged_at = Some(now); + inc.acknowledged_by = Some(actor.to_owned()); + } + inc.status = IncidentStatus::Resolved; + inc.resolved_at = Some(now); + inc.resolved_by = Some(actor.to_owned()); + inc.push(now, "resolved", actor, note.unwrap_or_else(|| "incident resolved".into())); + histogram!("incident_time_to_resolve_seconds", "severity" => inc.severity.as_str()) + .record(now.saturating_sub(inc.created_at) as f64); + if inc.severity.requires_postmortem() { + inc.push(now, "postmortem_required", "system", "SEV1/SEV2: postmortem due within 5 business days"); + } + if let Some(fp) = inc.alert_fingerprint { + alerting::ALERT_MANAGER.resolve(fp); + } + open.remove(&inc.dedup_key); + self.notify(ProviderAction::Resolve, inc, json!({}), actor); + Ok(inc.clone()) + } + + /// Resolve by dedup key (used when the upstream alert clears). + pub fn resolve_by_key(&self, dedup_key: &str, actor: &str, now: u64) -> Option { + let id = { self.open_by_key.lock().unwrap().get(dedup_key).cloned()? }; + self.resolve(&id, actor, Some("source alert resolved".into()), now).ok() + } + + pub fn add_note(&self, id: &str, actor: &str, message: String, now: u64) -> Result { + let mut incidents = self.incidents.lock().unwrap(); + let inc = incidents.get_mut(id).ok_or(IncidentError::NotFound)?; + inc.push(now, "note", actor, message); + Ok(inc.clone()) + } + + pub fn submit_postmortem(&self, id: &str, mut pm: Postmortem, now: u64) -> Result { + let mut incidents = self.incidents.lock().unwrap(); + let inc = incidents.get_mut(id).ok_or(IncidentError::NotFound)?; + if inc.status != IncidentStatus::Resolved { + return Err(IncidentError::InvalidState("postmortem can only be attached to a resolved incident")); + } + pm.submitted_at = now; + let author = pm.author.clone(); + inc.postmortem = Some(pm); + inc.push(now, "postmortem", &author, "postmortem submitted"); + counter!("incident_postmortems_total").increment(1); + Ok(inc.clone()) + } + + pub fn get(&self, id: &str) -> Option { + self.incidents.lock().unwrap().get(id).cloned() + } + + pub fn list(&self, status: Option) -> Vec { + let mut out: Vec = self + .incidents + .lock() + .unwrap() + .values() + .filter(|i| status.map_or(true, |s| i.status == s)) + .cloned() + .collect(); + out.sort_by(|a, b| b.created_at.cmp(&a.created_at)); + out + } + + pub fn stats(&self) -> IncidentStats { + let incidents = self.incidents.lock().unwrap(); + let mut s = IncidentStats { provider: self.provider.name().into(), ..Default::default() }; + let (mut tta, mut ttr) = (Vec::new(), Vec::new()); + for i in incidents.values() { + match i.status { + IncidentStatus::Triggered => s.triggered += 1, + IncidentStatus::Acknowledged => s.acknowledged += 1, + IncidentStatus::Resolved => s.resolved += 1, + } + if i.postmortem_due() { + s.postmortems_due.push(i.id.clone()); + } + if let Some(a) = i.acknowledged_at { + tta.push(a.saturating_sub(i.created_at)); + } + if let Some(r) = i.resolved_at { + ttr.push(r.saturating_sub(i.created_at)); + } + } + let mean = |v: &[u64]| if v.is_empty() { None } else { Some(v.iter().sum::() / v.len() as u64) }; + s.mean_time_to_acknowledge_seconds = mean(&tta); + s.mean_time_to_resolve_seconds = mean(&ttr); + s + } + + /// Fire-and-forget delivery to the provider. Failures are logged and + /// counted but never block the caller; the incident is still tracked + /// locally and Alertmanager's own PagerDuty route remains as a backstop. + fn notify(&self, action: ProviderAction, incident: &Incident, details: serde_json::Value, actor: &str) { + if !incident.paged { + return; + } + let Some(req) = self.provider.request(&self.client, action, incident, &details, actor) else { + return; + }; + let provider = self.provider.name(); + let id = incident.id.clone(); + let Ok(handle) = tokio::runtime::Handle::try_current() else { + tracing::warn!(incident = %id, "no tokio runtime; skipping {provider} delivery"); + return; + }; + handle.spawn(async move { + let result = req.send().await.and_then(|r| r.error_for_status()); + let outcome = if result.is_ok() { "ok" } else { "error" }; + counter!("incident_provider_requests_total", "provider" => provider, "outcome" => outcome).increment(1); + if let Err(e) = result { + tracing::error!(incident = %id, provider, action = ?action, error = %e, "incident provider request failed"); + } + }); + } +} + +#[derive(Debug, Default, Serialize)] +pub struct IncidentStats { + pub provider: String, + pub triggered: usize, + pub acknowledged: usize, + pub resolved: usize, + pub mean_time_to_acknowledge_seconds: Option, + pub mean_time_to_resolve_seconds: Option, + pub postmortems_due: Vec, +} + +/// Process-wide incident manager. +pub static INCIDENTS: Lazy = Lazy::new(|| IncidentManager::new(Provider::from_env())); + +// ── HTTP handlers ──────────────────────────────────────────────────────────── + +type ApiResult = Result, (StatusCode, Json)>; + +fn map_err(e: IncidentError) -> (StatusCode, Json) { + match e { + IncidentError::NotFound => (StatusCode::NOT_FOUND, Json(json!({ "error": "incident not found" }))), + IncidentError::InvalidState(msg) => (StatusCode::CONFLICT, Json(json!({ "error": msg }))), + } +} + +#[derive(Debug, Deserialize)] +pub struct ListQuery { + pub status: Option, +} + +/// `GET /v1/admin/incidents?status=triggered|acknowledged|resolved` +pub async fn list_handler(axum::extract::Query(q): axum::extract::Query) -> Json> { + Json(INCIDENTS.list(q.status)) +} + +/// `GET /v1/admin/incidents/stats` – counts, MTTA/MTTR, overdue postmortems. +pub async fn stats_handler() -> Json { + Json(INCIDENTS.stats()) +} + +/// `GET /v1/admin/incidents/{id}` +pub async fn get_handler(Path(id): Path) -> ApiResult { + INCIDENTS.get(&id).map(Json).ok_or_else(|| map_err(IncidentError::NotFound)) +} + +#[derive(Debug, Deserialize)] +pub struct CreateRequest { + pub title: String, + pub severity: IncidentSeverity, + pub reporter: String, + #[serde(default)] + pub description: Option, +} + +/// `POST /v1/admin/incidents` – declare an incident manually. +pub async fn create_handler(Json(req): Json) -> (StatusCode, Json) { + let now = alerting::now_secs(); + let (id, _) = INCIDENTS.trigger( + NewIncident { + dedup_key: format!("manual-{}", uuid::Uuid::new_v4().simple()), + title: req.title, + severity: req.severity, + source: format!("manual:{}", req.reporter), + alert_fingerprint: None, + paged: true, + details: json!({ "reporter": req.reporter, "description": req.description }), + }, + now, + ); + (StatusCode::CREATED, Json(INCIDENTS.get(&id).expect("just inserted"))) +} + +#[derive(Debug, Deserialize)] +pub struct ActorRequest { + pub actor: String, + #[serde(default)] + pub note: Option, +} + +/// `POST /v1/admin/incidents/{id}/acknowledge` +pub async fn acknowledge_handler(Path(id): Path, Json(req): Json) -> ApiResult { + INCIDENTS.acknowledge(&id, &req.actor, alerting::now_secs()).map(Json).map_err(map_err) +} + +/// `POST /v1/admin/incidents/{id}/resolve` +pub async fn resolve_handler(Path(id): Path, Json(req): Json) -> ApiResult { + INCIDENTS.resolve(&id, &req.actor, req.note, alerting::now_secs()).map(Json).map_err(map_err) +} + +#[derive(Debug, Deserialize)] +pub struct NoteRequest { + pub actor: String, + pub message: String, +} + +/// `POST /v1/admin/incidents/{id}/notes` – append to the incident timeline. +pub async fn note_handler(Path(id): Path, Json(req): Json) -> ApiResult { + INCIDENTS.add_note(&id, &req.actor, req.message, alerting::now_secs()).map(Json).map_err(map_err) +} + +/// `POST /v1/admin/incidents/{id}/postmortem` +pub async fn postmortem_handler(Path(id): Path, Json(pm): Json) -> ApiResult { + INCIDENTS.submit_postmortem(&id, pm, alerting::now_secs()).map(Json).map_err(map_err) +} + +/// Subset of the Alertmanager webhook payload (version 4). +#[derive(Debug, Deserialize)] +pub struct AlertmanagerWebhook { + pub alerts: Vec, +} + +#[derive(Debug, Deserialize)] +pub struct AlertmanagerAlert { + pub status: String, + #[serde(default)] + pub labels: HashMap, + #[serde(default)] + pub annotations: HashMap, + #[serde(default)] + pub fingerprint: String, + #[serde(default, rename = "generatorURL")] + pub generator_url: String, +} + +/// `POST /v1/admin/incidents/alertmanager` – Alertmanager webhook receiver. +/// Firing alerts open (or re-trigger) incidents; resolved alerts resolve them. +pub async fn alertmanager_webhook_handler(Json(hook): Json) -> Json { + let now = alerting::now_secs(); + let (mut opened, mut resolved) = (Vec::new(), Vec::new()); + for a in hook.alerts { + let name = a.labels.get("alertname").cloned().unwrap_or_else(|| "UnknownAlert".into()); + let dedup_key = format!("am-{}", if a.fingerprint.is_empty() { &name } else { &a.fingerprint }); + if a.status == "resolved" { + if let Some(inc) = INCIDENTS.resolve_by_key(&dedup_key, "alertmanager", now) { + resolved.push(inc.id); + } + continue; + } + let summary = a.annotations.get("summary").cloned().unwrap_or_default(); + let (id, _) = INCIDENTS.trigger( + NewIncident { + dedup_key, + title: if summary.is_empty() { name.clone() } else { format!("{name}: {summary}") }, + severity: IncidentSeverity::from_label(a.labels.get("severity").map(String::as_str)), + source: "alertmanager".into(), + alert_fingerprint: None, + paged: false, + details: json!({ + "labels": a.labels, + "annotations": a.annotations, + "generator_url": a.generator_url, + }), + }, + now, + ); + opened.push(id); + } + Json(json!({ "opened_or_retriggered": opened, "resolved": resolved })) +} diff --git a/api-server/src/main.rs b/api-server/src/main.rs index 97e19de..053e7b7 100644 --- a/api-server/src/main.rs +++ b/api-server/src/main.rs @@ -39,6 +39,7 @@ impl FromRef for Arc { } mod alerting; +mod incidents; mod auth; mod auth_2fa; mod session; @@ -308,6 +309,15 @@ async fn main() { .route("/v1/admin/audit/suspicious-patterns", get(handlers::get_suspicious_patterns)) .route("/v1/admin/commitments/lifecycle", get(commitment_monitoring::lifecycle_handler)) .route("/v1/admin/alerts", get(commitment_monitoring::open_alerts_handler)) + // #1068: incident management (PagerDuty / Opsgenie). + .route("/v1/admin/incidents", get(incidents::list_handler).post(incidents::create_handler)) + .route("/v1/admin/incidents/stats", get(incidents::stats_handler)) + .route("/v1/admin/incidents/alertmanager", post(incidents::alertmanager_webhook_handler)) + .route("/v1/admin/incidents/{id}", get(incidents::get_handler)) + .route("/v1/admin/incidents/{id}/acknowledge", post(incidents::acknowledge_handler)) + .route("/v1/admin/incidents/{id}/resolve", post(incidents::resolve_handler)) + .route("/v1/admin/incidents/{id}/notes", post(incidents::note_handler)) + .route("/v1/admin/incidents/{id}/postmortem", post(incidents::postmortem_handler)) .route("/v1/auth/recovery/initiate", post(account_recovery::initiate_recovery)) .route("/v1/auth/recovery/verify-token", post(account_recovery::verify_recovery_token)) .route("/v1/auth/recovery/questions", get(account_recovery::get_security_questions)) @@ -428,6 +438,15 @@ fn build_app() -> Router { .route("/v1/admin/audit/suspicious-patterns", get(handlers::get_suspicious_patterns)) .route("/v1/admin/commitments/lifecycle", get(commitment_monitoring::lifecycle_handler)) .route("/v1/admin/alerts", get(commitment_monitoring::open_alerts_handler)) + // #1068: incident management (PagerDuty / Opsgenie). + .route("/v1/admin/incidents", get(incidents::list_handler).post(incidents::create_handler)) + .route("/v1/admin/incidents/stats", get(incidents::stats_handler)) + .route("/v1/admin/incidents/alertmanager", post(incidents::alertmanager_webhook_handler)) + .route("/v1/admin/incidents/{id}", get(incidents::get_handler)) + .route("/v1/admin/incidents/{id}/acknowledge", post(incidents::acknowledge_handler)) + .route("/v1/admin/incidents/{id}/resolve", post(incidents::resolve_handler)) + .route("/v1/admin/incidents/{id}/notes", post(incidents::note_handler)) + .route("/v1/admin/incidents/{id}/postmortem", post(incidents::postmortem_handler)) .route("/v1/auth/recovery/initiate", post(account_recovery::initiate_recovery)) .route("/v1/auth/recovery/verify-token", post(account_recovery::verify_recovery_token)) .route("/v1/auth/recovery/questions", get(account_recovery::get_security_questions)) diff --git a/deploy/argocd/applications.yaml b/deploy/argocd/applications.yaml new file mode 100644 index 0000000..2c343d4 --- /dev/null +++ b/deploy/argocd/applications.yaml @@ -0,0 +1,70 @@ +# #1069: Argo CD Applications. Bootstrap once with +# kubectl apply -n argocd -f deploy/argocd/ +# after which Argo CD continuously reconciles the cluster to Git. +apiVersion: argoproj.io/v1alpha1 +kind: Application +metadata: + name: atomicip-staging + namespace: argocd + finalizers: + - resources-finalizer.argocd.argoproj.io +spec: + project: atomicip + source: + repoURL: https://github.com/AtomicIP/AtomicIP-.git + targetRevision: main + path: deploy/k8s/overlays/staging + destination: + server: https://kubernetes.default.svc + namespace: atomicip-staging + syncPolicy: + automated: + prune: true + selfHeal: true + syncOptions: + - CreateNamespace=true + - ApplyOutOfSyncOnly=true + retry: + limit: 5 + backoff: { duration: 10s, factor: 2, maxDuration: 3m } +--- +apiVersion: argoproj.io/v1alpha1 +kind: Application +metadata: + name: atomicip-production + namespace: argocd + finalizers: + - resources-finalizer.argocd.argoproj.io + annotations: + # Sync outcomes are forwarded to the incident pipeline (#1068). + notifications.argoproj.io/subscribe.on-sync-failed.pagerduty: atomicip-production + notifications.argoproj.io/subscribe.on-health-degraded.pagerduty: atomicip-production + notifications.argoproj.io/subscribe.on-deployed.slack: atomicip-deploys +spec: + project: atomicip + source: + repoURL: https://github.com/AtomicIP/AtomicIP-.git + targetRevision: main + path: deploy/k8s/overlays/production + destination: + server: https://kubernetes.default.svc + namespace: atomicip-production + # Automated sync from Git: whatever is merged to main in the production + # overlay is what runs. selfHeal reverts manual drift in the cluster. + syncPolicy: + automated: + prune: true + selfHeal: true + syncOptions: + - CreateNamespace=true + - ApplyOutOfSyncOnly=true + - PruneLast=true + retry: + limit: 3 + backoff: { duration: 30s, factor: 2, maxDuration: 5m } + # The HPA owns replica counts (#1067). + ignoreDifferences: + - group: apps + kind: Deployment + jsonPointers: + - /spec/replicas diff --git a/deploy/argocd/argocd-cm.yaml b/deploy/argocd/argocd-cm.yaml new file mode 100644 index 0000000..ea0d6d7 --- /dev/null +++ b/deploy/argocd/argocd-cm.yaml @@ -0,0 +1,15 @@ +# #1069: Argo CD settings required by the AtomicIP manifests. +# Merged into the argocd-cm ConfigMap installed by Argo CD. +apiVersion: v1 +kind: ConfigMap +metadata: + name: argocd-cm + namespace: argocd + labels: + app.kubernetes.io/name: argocd-cm + app.kubernetes.io/part-of: argocd +data: + # deploy/k8s/jobs packages scripts/ops/*.sh from outside its directory. + kustomize.buildOptions: --load-restrictor LoadRestrictionsNone + # Poll Git every 3 minutes (webhook from GitHub makes this near-instant). + timeout.reconciliation: 180s diff --git a/deploy/argocd/project.yaml b/deploy/argocd/project.yaml new file mode 100644 index 0000000..9aa98bb --- /dev/null +++ b/deploy/argocd/project.yaml @@ -0,0 +1,33 @@ +# #1069: Argo CD project scoping what the AtomicIP applications may deploy. +apiVersion: argoproj.io/v1alpha1 +kind: AppProject +metadata: + name: atomicip + namespace: argocd +spec: + description: AtomicIP api-server and operational jobs + sourceRepos: + - https://github.com/AtomicIP/AtomicIP-.git + destinations: + - server: https://kubernetes.default.svc + namespace: atomicip-staging + - server: https://kubernetes.default.svc + namespace: atomicip-production + clusterResourceWhitelist: [] + namespaceResourceWhitelist: + - { group: "", kind: ConfigMap } + - { group: "", kind: Service } + - { group: "", kind: ServiceAccount } + - { group: "", kind: PersistentVolumeClaim } + - { group: apps, kind: Deployment } + - { group: autoscaling, kind: HorizontalPodAutoscaler } + - { group: policy, kind: PodDisruptionBudget } + - { group: batch, kind: CronJob } + # Production syncs are blocked outside business hours unless overridden + # manually by an on-call engineer in the Argo CD UI. + syncWindows: + - kind: allow + schedule: "0 8 * * 1-5" + duration: 10h + applications: ["atomicip-production"] + manualSync: true diff --git a/deploy/k8s/base/deployment.yaml b/deploy/k8s/base/deployment.yaml new file mode 100644 index 0000000..05db489 --- /dev/null +++ b/deploy/k8s/base/deployment.yaml @@ -0,0 +1,80 @@ +apiVersion: apps/v1 +kind: Deployment +metadata: + name: atomicip-api + annotations: + # Argo CD syncs this resource in wave 0, after config (-1) and before jobs (1). + argocd.argoproj.io/sync-wave: "0" +spec: + # replicas is owned by the HorizontalPodAutoscaler (#1067); leaving it unset + # prevents Argo CD from fighting the autoscaler on every sync. + revisionHistoryLimit: 5 + selector: + matchLabels: + app: atomicip-api + strategy: + type: RollingUpdate + rollingUpdate: + maxSurge: 25% + maxUnavailable: 0 + template: + metadata: + labels: + app: atomicip-api + annotations: + prometheus.io/scrape: "true" + prometheus.io/port: "8080" + prometheus.io/path: /metrics + spec: + serviceAccountName: atomicip-api + terminationGracePeriodSeconds: 30 + securityContext: + runAsNonRoot: true + runAsUser: 10001 + fsGroup: 10001 + seccompProfile: + type: RuntimeDefault + containers: + - name: api-server + image: ghcr.io/atomicip/api-server + imagePullPolicy: IfNotPresent + ports: + - name: http + containerPort: 8080 + envFrom: + - configMapRef: + name: atomicip-api-config + - secretRef: + name: atomicip-api-secrets + optional: true + # Right-sized requests are the main cost lever (#1067): requests + # drive scheduling/bin-packing, limits only cap bursts. + resources: + requests: + cpu: 100m + memory: 128Mi + limits: + cpu: "1" + memory: 512Mi + readinessProbe: + httpGet: + path: /health + port: http + periodSeconds: 10 + failureThreshold: 3 + livenessProbe: + httpGet: + path: /health + port: http + initialDelaySeconds: 15 + periodSeconds: 20 + securityContext: + allowPrivilegeEscalation: false + readOnlyRootFilesystem: true + capabilities: + drop: ["ALL"] +--- +apiVersion: v1 +kind: ServiceAccount +metadata: + name: atomicip-api diff --git a/deploy/k8s/base/hpa.yaml b/deploy/k8s/base/hpa.yaml new file mode 100644 index 0000000..cdd25b8 --- /dev/null +++ b/deploy/k8s/base/hpa.yaml @@ -0,0 +1,40 @@ +# #1067: Auto-scaling for the api-server. Scales out on CPU/memory pressure and +# scales in conservatively so idle capacity is released without flapping. +apiVersion: autoscaling/v2 +kind: HorizontalPodAutoscaler +metadata: + name: atomicip-api +spec: + scaleTargetRef: + apiVersion: apps/v1 + kind: Deployment + name: atomicip-api + minReplicas: 2 + maxReplicas: 10 + metrics: + - type: Resource + resource: + name: cpu + target: + type: Utilization + averageUtilization: 70 + - type: Resource + resource: + name: memory + target: + type: Utilization + averageUtilization: 80 + behavior: + scaleUp: + stabilizationWindowSeconds: 60 + policies: + - type: Percent + value: 100 + periodSeconds: 60 + scaleDown: + # Wait 5 minutes of sustained low load, then remove at most 1 pod/min. + stabilizationWindowSeconds: 300 + policies: + - type: Pods + value: 1 + periodSeconds: 60 diff --git a/deploy/k8s/base/kustomization.yaml b/deploy/k8s/base/kustomization.yaml new file mode 100644 index 0000000..5673efa --- /dev/null +++ b/deploy/k8s/base/kustomization.yaml @@ -0,0 +1,24 @@ +# #1069: Base Kubernetes manifests for the AtomicIP api-server. +# Environments are Kustomize overlays under ../overlays and are synced to the +# cluster by Argo CD (see ../../argocd and docs/gitops.md). Never `kubectl +# apply` these by hand in production: Git is the source of truth. +apiVersion: kustomize.config.k8s.io/v1beta1 +kind: Kustomization + +labels: + - pairs: + app.kubernetes.io/name: atomicip-api + app.kubernetes.io/part-of: atomicip + app.kubernetes.io/managed-by: argocd + includeSelectors: false + +resources: + - deployment.yaml + - service.yaml + - hpa.yaml + - pdb.yaml + - ../jobs + +images: + - name: ghcr.io/atomicip/api-server + newTag: latest diff --git a/deploy/k8s/base/pdb.yaml b/deploy/k8s/base/pdb.yaml new file mode 100644 index 0000000..1f2d922 --- /dev/null +++ b/deploy/k8s/base/pdb.yaml @@ -0,0 +1,10 @@ +# Keeps at least one pod serving while the HPA or cluster autoscaler drains nodes. +apiVersion: policy/v1 +kind: PodDisruptionBudget +metadata: + name: atomicip-api +spec: + minAvailable: 1 + selector: + matchLabels: + app: atomicip-api diff --git a/deploy/k8s/base/service.yaml b/deploy/k8s/base/service.yaml new file mode 100644 index 0000000..9cd870e --- /dev/null +++ b/deploy/k8s/base/service.yaml @@ -0,0 +1,11 @@ +apiVersion: v1 +kind: Service +metadata: + name: atomicip-api +spec: + selector: + app: atomicip-api + ports: + - name: http + port: 80 + targetPort: http diff --git a/deploy/k8s/jobs/backup-verification-cronjob.yaml b/deploy/k8s/jobs/backup-verification-cronjob.yaml new file mode 100644 index 0000000..8200cfe --- /dev/null +++ b/deploy/k8s/jobs/backup-verification-cronjob.yaml @@ -0,0 +1,52 @@ +# #1066: Daily restore test of the latest backup. Needs a Docker-capable runner +# (dind sidecar) because the restore happens in a throwaway Postgres container. +apiVersion: batch/v1 +kind: CronJob +metadata: + name: backup-verification + annotations: + argocd.argoproj.io/sync-wave: "1" +spec: + schedule: "30 4 * * *" # after the 02:00 UTC base backup + concurrencyPolicy: Forbid + successfulJobsHistoryLimit: 3 + failedJobsHistoryLimit: 5 + jobTemplate: + spec: + backoffLimit: 0 + activeDeadlineSeconds: 10800 + template: + spec: + restartPolicy: Never + serviceAccountName: atomicip-backup-verifier + containers: + - name: verify + image: ghcr.io/atomicip/ops-toolbox:latest # aws-cli, jq, zstd, docker-cli + command: ["bash", "/scripts/backup-verify.sh"] + env: + - { name: DOCKER_HOST, value: "tcp://localhost:2375" } + - { name: RTO_TARGET_SECONDS, value: "7200" } + - { name: PUSHGATEWAY_URL, value: "http://prometheus-pushgateway.monitoring:9091" } + - { name: ALERTMANAGER_URL, value: "http://alertmanager.monitoring:9093" } + - { name: REPORT_OUT, value: "/tmp/backup-verification.json" } + - name: BACKUP_BUCKET + valueFrom: { secretKeyRef: { name: atomicip-backup, key: bucket } } + resources: + requests: { cpu: 250m, memory: 256Mi } + volumeMounts: + - { name: scripts, mountPath: /scripts } + - { name: work, mountPath: /tmp } + - name: dind + image: docker:27-dind + args: ["--tls=false"] + env: + - { name: DOCKER_TLS_CERTDIR, value: "" } + securityContext: + privileged: true + volumeMounts: + - { name: work, mountPath: /tmp } + volumes: + - name: scripts + configMap: { name: atomicip-ops-scripts, defaultMode: 0555 } + - name: work + emptyDir: { sizeLimit: 50Gi } diff --git a/deploy/k8s/jobs/cost-optimization-cronjob.yaml b/deploy/k8s/jobs/cost-optimization-cronjob.yaml new file mode 100644 index 0000000..3aabf01 --- /dev/null +++ b/deploy/k8s/jobs/cost-optimization-cronjob.yaml @@ -0,0 +1,45 @@ +# #1067: Nightly orphaned-data cleanup + compression; weekly report. +apiVersion: batch/v1 +kind: CronJob +metadata: + name: cost-optimization + annotations: + argocd.argoproj.io/sync-wave: "1" +spec: + schedule: "0 3 * * *" + concurrencyPolicy: Forbid + jobTemplate: + spec: + backoffLimit: 1 + template: + spec: + restartPolicy: Never + serviceAccountName: atomicip-cost-optimizer + containers: + - name: optimize + image: ghcr.io/atomicip/ops-toolbox:latest # aws-cli, psql, jq, zstd + command: ["bash", "/scripts/cost-optimize.sh", "all"] + env: + - { name: RETENTION_DAYS, value: "30" } + - { name: COMPRESS_AFTER_DAYS, value: "7" } + - { name: COLD_TIER_AFTER_DAYS, value: "90" } + - { name: COST_LEDGER, value: "/ledger/cost-ledger.jsonl" } + - { name: REPORT_OUT, value: "/ledger/cost-report-latest.md" } + - { name: PUSHGATEWAY_URL, value: "http://prometheus-pushgateway.monitoring:9091" } + - name: DATABASE_URL + valueFrom: { secretKeyRef: { name: atomicip-api-secrets, key: DATABASE_URL } } + - name: ARCHIVE_BUCKET + valueFrom: { secretKeyRef: { name: atomicip-backup, key: archive_bucket } } + resources: + requests: { cpu: 100m, memory: 128Mi } + volumeMounts: + - { name: scripts, mountPath: /scripts } + - { name: ledger, mountPath: /ledger } + - { name: audit, mountPath: /var/log/atomicip/audit } + volumes: + - name: scripts + configMap: { name: atomicip-ops-scripts, defaultMode: 0555 } + - name: ledger + persistentVolumeClaim: { claimName: atomicip-cost-ledger } + - name: audit + persistentVolumeClaim: { claimName: atomicip-audit-logs } diff --git a/deploy/k8s/jobs/kustomization.yaml b/deploy/k8s/jobs/kustomization.yaml new file mode 100644 index 0000000..b5d9f1b --- /dev/null +++ b/deploy/k8s/jobs/kustomization.yaml @@ -0,0 +1,17 @@ +# Scheduled operational jobs (#1066 backup verification, #1067 cost optimization). +# The scripts are shipped as a ConfigMap so the jobs always run the version in Git. +# Referencing ../../../scripts requires `--load-restrictor LoadRestrictionsNone` +# (set in deploy/argocd/argocd-cm.yaml and in gitops-validate.yml). +apiVersion: kustomize.config.k8s.io/v1beta1 +kind: Kustomization + +resources: + - rbac-storage.yaml + - backup-verification-cronjob.yaml + - cost-optimization-cronjob.yaml + +configMapGenerator: + - name: atomicip-ops-scripts + files: + - backup-verify.sh=../../../scripts/ops/backup-verify.sh + - cost-optimize.sh=../../../scripts/ops/cost-optimize.sh diff --git a/deploy/k8s/jobs/rbac-storage.yaml b/deploy/k8s/jobs/rbac-storage.yaml new file mode 100644 index 0000000..c1d14f3 --- /dev/null +++ b/deploy/k8s/jobs/rbac-storage.yaml @@ -0,0 +1,20 @@ +apiVersion: v1 +kind: ServiceAccount +metadata: + name: atomicip-backup-verifier + # Bind to a read-only role on the backup bucket (IRSA / Workload Identity). +--- +apiVersion: v1 +kind: ServiceAccount +metadata: + name: atomicip-cost-optimizer +--- +apiVersion: v1 +kind: PersistentVolumeClaim +metadata: + name: atomicip-cost-ledger +spec: + accessModes: ["ReadWriteOnce"] + resources: + requests: + storage: 1Gi diff --git a/deploy/k8s/overlays/production/kustomization.yaml b/deploy/k8s/overlays/production/kustomization.yaml new file mode 100644 index 0000000..a33e71a --- /dev/null +++ b/deploy/k8s/overlays/production/kustomization.yaml @@ -0,0 +1,34 @@ +# #1069: Production environment. Changes land ONLY via a reviewed pull request +# (see docs/gitops.md). Rollback = `git revert` of the promotion commit. +apiVersion: kustomize.config.k8s.io/v1beta1 +kind: Kustomization + +namespace: atomicip-production + +resources: + - ../../base + +configMapGenerator: + - name: atomicip-api-config + literals: + - RUST_LOG=warn + - STELLAR_NETWORK=mainnet + - PORT=8080 + - INCIDENT_PROVIDER=pagerduty + +patches: + - target: + kind: HorizontalPodAutoscaler + name: atomicip-api + patch: | + - op: replace + path: /spec/minReplicas + value: 3 + - op: replace + path: /spec/maxReplicas + value: 20 + +# Updated by .github/workflows/gitops-promote.yml via pull request — do not edit by hand. +images: + - name: ghcr.io/atomicip/api-server + newTag: latest diff --git a/deploy/k8s/overlays/staging/kustomization.yaml b/deploy/k8s/overlays/staging/kustomization.yaml new file mode 100644 index 0000000..fd3847c --- /dev/null +++ b/deploy/k8s/overlays/staging/kustomization.yaml @@ -0,0 +1,33 @@ +# #1069: Staging environment. Auto-synced by Argo CD on every merge to main. +apiVersion: kustomize.config.k8s.io/v1beta1 +kind: Kustomization + +namespace: atomicip-staging + +resources: + - ../../base + +configMapGenerator: + - name: atomicip-api-config + literals: + - RUST_LOG=info + - STELLAR_NETWORK=testnet + - PORT=8080 + - INCIDENT_PROVIDER=none + +patches: + - target: + kind: HorizontalPodAutoscaler + name: atomicip-api + patch: | + - op: replace + path: /spec/minReplicas + value: 1 + - op: replace + path: /spec/maxReplicas + value: 3 + +# Updated by .github/workflows/gitops-promote.yml — do not edit by hand. +images: + - name: ghcr.io/atomicip/api-server + newTag: latest diff --git a/docs/backup-strategy.md b/docs/backup-strategy.md new file mode 100644 index 0000000..aa40291 --- /dev/null +++ b/docs/backup-strategy.md @@ -0,0 +1,105 @@ +# Backup Strategy and Verification + +> Issue #1066. Owner: Platform on-call. Backup schedule and RTO/RPO targets are +> defined in [disaster-recovery.md](disaster-recovery.md) §2–3; this page +> covers how we **prove** those backups are restorable. + +A backup that has never been restored is a hope, not a backup. Every day the +latest backup is restored into a throwaway Postgres and checked end to end. + +## 1. What is backed up + +| Asset | Method | Frequency | Retention | +|-------|--------|-----------|-----------| +| Postgres | Base backup (`pg_basebackup`, with `backup_manifest` + `SHA256SUMS`) + continuous WAL archive | Base daily 02:00 UTC, WAL continuous | 35 days PITR, monthly 1 year | +| Audit log | Append-only sync to object-locked bucket (zstd) | 15 min | 7 years | +| Secrets, admin keys | Secret-manager replication; Shamir 3-of-5 | On change | Permanent | +| Contract state | Stellar ledger (not backed up by us) | — | — | +| Deployment state | Git (`deploy/`, GitOps #1069) | Every change | Permanent | + +Expected bucket layout (`BACKUP_BUCKET`): + +``` +postgres/base//{base.tar.gz, pg_wal.tar.gz, backup_manifest, SHA256SUMS} +postgres/wal/ +audit///
/*.log.zst +``` + +Encryption: SSE-KMS at rest, cross-region replication to the DR region. + +## 2. Verification pipeline + +`scripts/ops/backup-verify.sh` runs: + +| Step | Check | Fails when | +|------|-------|------------| +| Freshness | Age of newest base backup, WAL segment and audit archive | Base > 26 h, WAL > 5 min, audit > 30 min | +| Integrity | `sha256sum -c SHA256SUMS`, `pg_verifybackup` against the manifest | Any checksum / manifest mismatch | +| Restore | Base backup + WAL replay (`recovery_target_timeline=latest`) in a disposable container | Postgres exits, or recovery exceeds RTO | +| Validation | Expected tables exist, row counts recorded, `amcheck` on every btree index | Missing table, index corruption | +| Audit archive | Newest archive decompresses (`zstd -t` / `gzip -t`) | Corrupt archive | +| RTO | Restore duration vs `RTO_TARGET_SECONDS` (2 h) | Slower than target | + +Where it runs: + +* **Daily, in-cluster**: `deploy/k8s/jobs/backup-verification-cronjob.yaml` at 04:30 UTC. +* **Weekly, off-cluster**: `.github/workflows/backup-verification.yml` (Monday 06:00 UTC) + restores from a completely separate environment, proving recoverability even + if the production cluster is lost. It uploads the JSON report (kept 90 days) + and opens a GitHub issue on failure. + +## 3. Metrics + +Pushed to the Pushgateway (`job="backup_verification"`): + +* `atomicip_backup_verify_success` (1/0), `atomicip_backup_verify_duration_seconds` +* `atomicip_backup_verify_last_success_timestamp_seconds` +* `atomicip_backup_restore_seconds` — measured recovery time +* `atomicip_backup_rpo_seconds` — gap between run start and last committed transaction in the restore +* `atomicip_backup_age_seconds{kind}`, `atomicip_backup_restored_rows{table}` + +Alerts (`monitoring/prometheus/operations-rules.yml`): `BackupVerificationFailed` +(critical), `BackupVerificationStale` (critical, no success in 48 h), `BackupTooOld` +(warning), `BackupRestoreSlowerThanRTO` (warning). On failure the script also +posts `BackupVerificationFailed` directly to Alertmanager, so an alert fires even +if the Pushgateway is down; critical alerts open an incident (#1068). + +## 4. Recovery time + +Track `atomicip_backup_restore_seconds` over time on the operations dashboard. +If it trends toward the 2 h RTO: + +1. Take base backups more often (fewer WAL segments to replay). +2. Restore from the same region as the backup bucket; use a larger restore instance. +3. Use parallel WAL prefetch (`pgBackRest`/`wal-g` with `--prefetch`). + +## 5. When verification fails + +1. The alert opens an incident (SEV2). Acknowledge it + ([incident-response.md](incident-response.md)). +2. Read the report (`backup-verification.json` in the workflow artifact or the CronJob logs: + `kubectl -n atomicip-production logs job/ -c verify`). +3. By failure type: + * **Stale backup** – check the backup job and WAL archiver (`pg_stat_archiver.failed_count`). + * **Checksum / manifest** – mark that backup bad; verify the previous one with + `BACKUP_BUCKET=… scripts/ops/backup-verify.sh` pointed at it; take a fresh base backup now. + * **Restore / amcheck** – treat as possible corruption in production; run `amcheck` on the primary. + * **RTO exceeded** – see §4; not an immediate data-loss risk. +4. Pause destructive jobs (cost cleanup, #1067) until a verification passes. +5. Close the incident only after a green verification run. + +## 6. Manual run + +```bash +BACKUP_BUCKET=s3://atomicip-backups/prod \ +PG_IMAGE=postgres:16 KEEP_WORKDIR=1 \ +scripts/ops/backup-verify.sh +``` + +Requirements: `aws` CLI with read access to the bucket, Docker, `jq`, `zstd`. + +## 7. Review + +Backup strategy is reviewed quarterly with the DR plan. Each quarterly DR drill +([disaster-recovery.md](disaster-recovery.md)) uses the verification output as its +starting point. diff --git a/docs/cost-optimization.md b/docs/cost-optimization.md new file mode 100644 index 0000000..3f2893c --- /dev/null +++ b/docs/cost-optimization.md @@ -0,0 +1,81 @@ +# Cost Optimization Automation + +> Issue #1067. Owner: Platform. Everything here is automated; this page explains +> what runs, what it touches and how savings are reported. + +| Lever | Implementation | Schedule | +|-------|----------------|----------| +| Auto-scaling | `deploy/k8s/base/hpa.yaml` | Continuous | +| Orphaned data cleanup | `scripts/ops/cost-optimize.sh cleanup` | Nightly 03:00 UTC (CronJob) | +| Compression & storage tiering | `scripts/ops/cost-optimize.sh compress` | Nightly 03:00 UTC (CronJob) | +| Savings tracking | JSONL ledger + Pushgateway metrics | Every run | +| Cost report | `.github/workflows/cost-report.yml` | Weekly, Monday 07:00 UTC | + +## 1. Auto-scaling + +The api-server is stateless (see [disaster-recovery.md](disaster-recovery.md) §1), +so it scales horizontally with a `HorizontalPodAutoscaler`: + +* Targets: 70 % CPU, 80 % memory of **requests** (100m CPU / 128Mi). +* Scale-up: up to +100 % per minute after a 60 s window. +* Scale-down: after 5 min of low load, at most one pod per minute (no flapping). +* Bounds: staging 1–3, production 3–20 (set in each overlay; change via PR, see [gitops.md](gitops.md)). +* A `PodDisruptionBudget` keeps one pod serving during node scale-down, letting + the cluster autoscaler remove idle nodes safely. +* `ApiServerAtMaxReplicas` fires if the HPA is pinned at its ceiling for 30 min. + +Right-sizing: review `atomicip:api_replicas:avg1d` and container CPU/memory +usage monthly; lower `requests` when p95 usage stays below 50 % of them. + +## 2. Orphaned data cleanup + +Only off-chain caches/indexes are cleaned; the Stellar ledger is never touched. + +| Target | Rule | +|--------|------| +| `expired_sessions` | `sessions.expires_at` older than `RETENTION_DAYS` (30) | +| `orphaned_webhook_deliveries` | Deliveries whose webhook was unregistered | +| `stale_webhook_deliveries` | Delivered more than `RETENTION_DAYS` ago | +| `expired_recovery_tokens` | Expired > 1 day ago | +| `orphaned_watchlist_entries` | Watchlist rows for IPs no longer indexed | +| `stale_exports` | `GET /ip/export` files in object storage older than 30 days | + +Rows are deleted in batches of 5 000, followed by `VACUUM (ANALYZE)`. Tables that +do not exist are skipped. Add rules to `ORPHAN_RULES` in the script. + +## 3. Compression of old data + +* **Audit logs**: rotated files older than `COMPRESS_AFTER_DAYS` (7) are + compressed with `zstd -19` (typically 8–12× for JSON logs) and integrity-checked + before the original is removed. The active log is never touched; the HMAC chain + is preserved byte-for-byte inside the archive. +* **Postgres partitions** of append-only tables (`audit_events`, `chain_events`) + older than a month switch to `lz4` TOAST compression and are rewritten. +* **Object storage tiering**: audit archives move to Standard-IA after + `COLD_TIER_AFTER_DAYS` (90) and Glacier Instant Retrieval after 360 days; + exports expire after 30 days. Object Lock retention is unaffected. + +## 4. Savings tracking and reports + +Each action appends to the ledger (`COST_LEDGER`, persisted on a PVC and synced to object storage): + +```json +{"ts":"2026-09-27T03:00:12Z","action":"cleanup","target":"expired_sessions","items":18234,"bytes_saved":41943040,"monthly_usd_saved":0.0009,"dry_run":false} +``` + +and pushes `atomicip_cost_bytes_reclaimed`, `atomicip_cost_monthly_usd_saved` +and `atomicip_cost_last_run_timestamp_seconds` to the Pushgateway. +`scripts/ops/cost-optimize.sh report` aggregates the ledger into a Markdown +table; the weekly workflow adds actual spend per service from AWS Cost Explorer +(resources tagged `project=atomicip`) and publishes it as a run summary + artifact. + +Savings estimates use `STORAGE_COST_PER_GB` / `COLD_COST_PER_GB`; update them when +provider pricing changes. Autoscaling savings are visible as replica-hours in +the `atomicip:api_replicas:avg1d` recording rule versus the fixed pre-HPA count. + +## 5. Operating safely + +* Always try `DRY_RUN=1 scripts/ops/cost-optimize.sh all` after changing rules. +* Cleanup never runs against a database that failed last night's backup + verification (check `BackupVerificationFailed` before re-running manually). +* `CostOptimizationJobStale` (info) fires when no run is recorded for 3 days. diff --git a/docs/disaster-recovery.md b/docs/disaster-recovery.md index 8852cae..ef43bc4 100644 --- a/docs/disaster-recovery.md +++ b/docs/disaster-recovery.md @@ -61,6 +61,9 @@ write endpoints (these depend on RPC). Every backup job emits `backup_last_success_timestamp_seconds{asset=...}`. Alert if this is older than 2× the backup interval. +Backups are restore-tested daily and weekly off-cluster; see +[backup-strategy.md](backup-strategy.md) (#1066). + ### 3.2 Restore: Postgres (point-in-time) 1. Declare an incident and freeze writes: scale the API server to read-only @@ -97,6 +100,9 @@ region or cluster, with the same environment (`SOROBAN_RPC_URL`, `GET /health/detailed` and run `scripts/smoke-test.sh` against the new endpoint before moving DNS or load-balancer traffic. +With GitOps (#1069) this means pointing Argo CD at the new cluster, or +reverting the bad promotion commit; see [gitops.md](gitops.md#rollback). + ### 3.5 Restore: Audit log Restore the log from the versioned bucket to `AUDIT_LOG_PATH`. Verify the HMAC chain @@ -126,6 +132,9 @@ path is `validate_upgrade` + `upgrade` with a new WASM hash. See §5.3. ## 5. Incident response playbooks +Incident tracking, paging integration (PagerDuty/Opsgenie) and postmortems: +[incident-response.md](incident-response.md) (#1068). + ### Roles and severity | Severity | Definition | Response | diff --git a/docs/gitops.md b/docs/gitops.md new file mode 100644 index 0000000..b616f0d --- /dev/null +++ b/docs/gitops.md @@ -0,0 +1,103 @@ +# GitOps Deployment Pipeline + +> Issue #1069. Owner: Platform. Tooling: [Argo CD](https://argo-cd.readthedocs.io/) + Kustomize. + +Deployments are declarative: the desired state of every environment lives in +Git under `deploy/`, and Argo CD continuously reconciles the cluster to it. +Nobody runs `kubectl apply` against production. + +## Layout + +``` +deploy/ +├── argocd/ +│ ├── project.yaml # AppProject: allowed repos, namespaces, kinds, sync windows +│ └── applications.yaml # atomicip-staging, atomicip-production Applications +└── k8s/ + ├── base/ # Deployment, Service, HPA (#1067), PDB + ├── jobs/ # CronJobs: backup verification (#1066), cost optimization (#1067) + └── overlays/ + ├── staging/ # namespace atomicip-staging, testnet, 1–3 replicas + └── production/ # namespace atomicip-production, mainnet, 3–20 replicas +``` + +## Bootstrap (once per cluster) + +```bash +kubectl create namespace argocd +kubectl apply -n argocd -f https://raw.githubusercontent.com/argoproj/argo-cd/v2.12.3/manifests/install.yaml +kubectl apply -n argocd -f deploy/argocd/ +``` + +`deploy/argocd/argocd-cm.yaml` sets `kustomize.buildOptions: --load-restrictor LoadRestrictionsNone` +so the jobs kustomization can package `scripts/ops/*.sh` into a ConfigMap. + +Create the out-of-band secrets (never committed): `atomicip-api-secrets` +(`DATABASE_URL`, `JWT_SECRET`, `PAGERDUTY_ROUTING_KEY`, …) and `atomicip-backup` +in each namespace, ideally via External Secrets / Sealed Secrets. + +## How a change reaches production + +```mermaid +flowchart LR + A[PR to api-server] -->|merge| B[gitops-promote: build image :sha] + B -->|commit tag| C[overlays/staging] + C -->|Argo CD auto-sync| D[staging cluster] + D -->|verified| E[workflow_dispatch tag=sha] + E -->|opens PR| F[overlays/production] + F -->|review + merge| G[Argo CD auto-sync] + G --> H[production cluster] +``` + +1. **Merge to `main`** (touching `api-server/`) → `.github/workflows/gitops-promote.yml` + builds `ghcr.io/atomicip/api-server:` and commits the new tag to + `deploy/k8s/overlays/staging`. Argo CD syncs staging automatically. +2. **Promote**: run *GitOps Promote* with `tag=`. The workflow verifies the + image exists and opens a pull request bumping the production overlay. +3. **Review**: `.github/workflows/gitops-validate.yml` renders both overlays, + validates them with kubeconform and posts the rendered diff on the PR. + Production changes need an approving review (protect `deploy/k8s/overlays/production/` + with CODEOWNERS + branch protection). +4. **Merge = deploy.** Argo CD (automated sync, `prune`, `selfHeal`) applies the + change. The `AppProject` sync window only allows automatic production syncs + Mon–Fri 08:00–18:00 UTC; on-call can sync manually outside it. + +Config changes (ConfigMap literals, HPA bounds, resource requests) follow the +same path: edit the overlay in a PR, merge, done. + +## Rollback + +Rollback is `git revert` of the promotion commit — Git history *is* the deploy log. + +```bash +scripts/gitops-rollback.sh production # revert newest production promotion +scripts/gitops-rollback.sh production # revert a specific commit +scripts/gitops-rollback.sh staging # staging: pushed directly to main +``` + +For production the script opens a revert PR; merge it (an expedited review is +fine during an incident) and Argo CD syncs the previous image. Emergency only: +`argocd app rollback atomicip-production ` restores the cluster +immediately, but **must be followed by the revert PR**, otherwise `selfHeal` +re-applies the bad revision on the next sync. + +Contract (Soroban) upgrades are *not* GitOps-managed; see +[contract-upgrades.md](contract-upgrades.md). + +## Drift and observability + +* `selfHeal: true` reverts manual edits in the cluster. Use `argocd app diff` to see drift. +* `ignoreDifferences` on `/spec/replicas` lets the HPA own replica counts. +* Argo CD notifications page on `sync-failed` / `health-degraded` for production + (incident pipeline, #1068) and post successful deploys to Slack. +* Prometheus alerts `ArgoAppOutOfSync` and `ArgoAppDegraded` live in + `monitoring/prometheus/operations-rules.yml`. + +## Troubleshooting + +| Symptom | Check | +|---------|-------| +| App `OutOfSync` for > 30 min | `argocd app get atomicip-production` — sync window closed? manifest error? | +| Sync fails with "resource not permitted" | Kind missing from `namespaceResourceWhitelist` in `project.yaml` | +| Promotion PR not created | Image tag missing in GHCR; check the `build-and-stage` run for that SHA | +| App `Degraded` after sync | Readiness probe failing — roll back first, investigate second | diff --git a/docs/incident-response.md b/docs/incident-response.md new file mode 100644 index 0000000..c4c55ae --- /dev/null +++ b/docs/incident-response.md @@ -0,0 +1,121 @@ +# Incident Response Procedures + +> Issue #1068. Owner: Platform on-call. Complements +> [disaster-recovery.md](disaster-recovery.md) (recovery steps) and +> [alerting-best-practices.md](alerting-best-practices.md) (alert design). + +## 1. Integration overview + +```mermaid +flowchart LR + P[Prometheus rules] --> AM[Alertmanager] + AP[api-server alert pipeline
alerting.rs] -->|critical| IM[incidents.rs] + AM -->|critical, webhook| IM + AM -->|critical| PD[PagerDuty / Opsgenie] + IM -->|in-process alerts & manual| PD + ACD[Argo CD notifications] --> PD +``` + +* **Automatic creation.** Every critical alert opens an incident: + in-process alerts from `api-server/src/alerting.rs` via the background + evaluator, and Prometheus alerts via the Alertmanager `incident-tracker` + webhook (`POST /v1/admin/incidents/alertmanager`). Re-fires of the same alert + are deduplicated onto the open incident (`trigger_count` + timeline entry); + when the alert resolves, the incident resolves. +* **Paging.** Set `INCIDENT_PROVIDER=pagerduty` (+ `PAGERDUTY_ROUTING_KEY`) or + `INCIDENT_PROVIDER=opsgenie` (+ `OPSGENIE_API_KEY`, optional `OPSGENIE_API_URL`). + Incidents from in-process alerts and manual declarations are sent to the + provider (PagerDuty Events API v2 / Opsgenie Alerts API). Alertmanager-sourced + incidents are tracked but *not* re-sent, because Alertmanager pages the + provider directly (`pagerduty-oncall` / `opsgenie-oncall` receivers). +* **Acknowledgment** in the API stops in-process escalation + (`ALERT_MANAGER.acknowledge`) and acknowledges at the provider. Acknowledging + in PagerDuty/Opsgenie directly is also fine; record it with the API so MTTA is tracked. + +### API + +| Method & path | Purpose | +|---------------|---------| +| `GET /v1/admin/incidents?status=` | List incidents (newest first) | +| `POST /v1/admin/incidents` | Declare manually: `{title, severity: "sev1"\|"sev2"\|"sev3", reporter, description?}` | +| `GET /v1/admin/incidents/{id}` | Incident with full timeline | +| `POST /v1/admin/incidents/{id}/acknowledge` | `{actor}` | +| `POST /v1/admin/incidents/{id}/notes` | `{actor, message}` – timeline entry | +| `POST /v1/admin/incidents/{id}/resolve` | `{actor, note?}` | +| `POST /v1/admin/incidents/{id}/postmortem` | Attach postmortem (resolved incidents only) | +| `GET /v1/admin/incidents/stats` | Open counts, MTTA, MTTR, overdue postmortems | + +Metrics: `incidents_created_total{severity}`, `incidents_retriggered_total`, +`incident_time_to_acknowledge_seconds`, `incident_time_to_resolve_seconds`, +`incident_provider_requests_total{provider,outcome}`, `incident_postmortems_total`. + +## 2. Severity levels + +| Severity | Definition | Examples | Ack target | Update cadence | +|----------|-----------|----------|-----------|----------------| +| **SEV1** | Funds or IP integrity at risk, or full outage | Escrow stuck, key compromise, API down for all users | 5 min | 30 min | +| **SEV2** | Major degradation, no data/funds at risk | Soroban RPC outage, write endpoints failing, backup verification failed | 15 min | 1 h | +| **SEV3** | Minor/partial degradation | Elevated latency, single non-critical job failing | 4 h (business hours) | Daily | + +Mapping: critical alert → SEV2; critical alert at escalation level ≥ 2 or +labelled `severity="sev1"` → SEV1; everything else → SEV3. Anyone may raise severity. + +## 3. Roles + +* **Incident Commander (IC)** – the first responder until handed over. Owns decisions, not keyboards. +* **Operations lead** – executes mitigation (rollback, failover, scaling). +* **Communications lead** – status page and stakeholder updates (SEV1/SEV2). +* **Scribe** – keeps the timeline (`/notes`) current. + +For SEV3 one person holds all roles. + +## 4. Response procedure + +1. **Acknowledge** within the target (PagerDuty/Opsgenie app or `POST …/acknowledge`). +2. **Assess**: confirm impact via Grafana, `/health/detailed`, `/v1/admin/alerts`. + Adjust severity if needed. Declare in `#atomicip-incidents`. +3. **Mitigate first, diagnose second.** Preferred mitigations, in order: + * Bad deploy → `scripts/gitops-rollback.sh production` ([gitops.md](gitops.md#rollback)). + * RPC outage → fail over provider ([disaster-recovery.md](disaster-recovery.md)). + * Load → confirm HPA is scaling; raise `maxReplicas` via overlay PR. + * Contract emergency → pause/freeze per [contract-upgrades.md](contract-upgrades.md). +4. **Communicate** on the cadence above. Every significant action goes into the timeline. +5. **Resolve** once impact has ended and monitoring is green for 15 min + (`POST …/resolve` with a one-line note). Alert-sourced incidents resolve automatically when the alert clears — verify, don't assume. +6. **Follow up**: SEV1/SEV2 require a postmortem (§6). + +## 5. Escalation + +If the primary does not acknowledge within the target, the provider escalation +policy pages the secondary, then the engineering manager. Escalate immediately to +security (`SECURITY.md`) for any suspected key compromise or exploit. + +## 6. Postmortems + +* Required for every SEV1/SEV2; due within **5 business days** of resolution. + `GET /v1/admin/incidents/stats` lists `postmortems_due`. +* Blameless: focus on systems and processes, not individuals. +* Write it with [postmortem-template.md](postmortem-template.md), review in the + weekly ops meeting, then attach it: + +```bash +curl -X POST "$API/v1/admin/incidents/INC-1A2B3C4D5E/postmortem" -H 'Content-Type: application/json' -d '{ + "author": "oncall-primary", + "summary": "Write endpoints returned 503 for 42 minutes", + "impact": "~1,200 failed commit requests; no funds or IP state affected", + "root_cause": "RPC provider rate-limited our key after a config change", + "contributing_factors": ["No alert on RPC 429 rate"], + "what_went_well": ["Rollback via git revert took 4 minutes"], + "what_went_wrong": ["Failover runbook referenced an old endpoint"], + "action_items": [{"description": "Alert on RPC 429s", "owner": "platform", "tracking_url": "https://github.com/AtomicIP/AtomicIP-/issues/NNN"}], + "document_url": "https://…/postmortems/2026-09-27-rpc-rate-limit.md" +}' +``` + +* Every action item gets a GitHub issue with the `postmortem` label. + +## 7. Drills + +Quarterly game day: fire a synthetic critical alert through Alertmanager +(`amtool alert add IncidentDrill severity=critical`), and run the procedure end +to end including a postmortem. Track MTTA/MTTR trends from `/v1/admin/incidents/stats`. diff --git a/docs/postmortem-template.md b/docs/postmortem-template.md new file mode 100644 index 0000000..4205b5a --- /dev/null +++ b/docs/postmortem-template.md @@ -0,0 +1,46 @@ +# Postmortem: + +> Blameless. Incident: `INC-XXXXXXXXXX` · Severity: SEV? · Date: YYYY-MM-DD · +> Author: <role> · Status: Draft / Reviewed + +## Summary + +Two or three sentences: what happened, how long, who was affected. + +## Impact + +* Duration (start → mitigated → resolved, UTC): +* Users / requests affected: +* Funds, IP records or audit trail affected? (yes/no + detail): +* SLO budget consumed: + +## Timeline (UTC) + +Export from `GET /v1/admin/incidents/{id}` and annotate. + +| Time | Event | +|------|-------| +| | Alert fired / incident triggered | +| | Acknowledged by | +| | Mitigation applied | +| | Resolved | + +## Root cause + +## Contributing factors + +## Detection + +How was it detected? Could it have been detected earlier? + +## What went well + +## What went wrong + +## Where we got lucky + +## Action items + +| Action | Type (prevent / detect / mitigate) | Owner | Issue | +|--------|-----------------------------------|-------|-------| +| | | | | diff --git a/monitoring/README.md b/monitoring/README.md index d6f6c73..2ea37f2 100644 --- a/monitoring/README.md +++ b/monitoring/README.md @@ -5,11 +5,15 @@ | `prometheus/commitment-lifecycle-rules.yml` | Recording + alert rules for commitment states, transition rates, anomalies and week-over-week trends | #1064 | | `grafana/commitment-lifecycle-dashboard.json` | Dashboard with an overview row, one row per state (active / revealed / archived), trends and alert-pipeline panels | #1064 | | `alertmanager/alertmanager.yml` | Grouping (dedup), inhibition (correlation), severity routing (escalation) | #1065 | +| `alertmanager/alertmanager.yml` (`incident-tracker`, `opsgenie-oncall`) | Critical alerts open incidents in the api-server; Opsgenie as alternative pager | #1068 | +| `prometheus/operations-rules.yml` | Backup verification, cost optimization / autoscaling, and Argo CD sync alerts | #1066, #1067, #1069 | Metrics are exported on the api-server `/metrics` endpoint by `api-server/src/commitment_monitoring.rs` and `api-server/src/alerting.rs`. The api-server also exposes JSON views at `GET /v1/admin/commitments/lifecycle` and `GET /v1/admin/alerts`. -See [`docs/alerting-best-practices.md`](../docs/alerting-best-practices.md) and -[`docs/disaster-recovery.md`](../docs/disaster-recovery.md). +See [`docs/alerting-best-practices.md`](../docs/alerting-best-practices.md), +[`docs/disaster-recovery.md`](../docs/disaster-recovery.md), +[`docs/incident-response.md`](../docs/incident-response.md) and +[`docs/backup-strategy.md`](../docs/backup-strategy.md). diff --git a/monitoring/alertmanager/alertmanager.yml b/monitoring/alertmanager/alertmanager.yml index d9f6dcb..72f31c0 100644 --- a/monitoring/alertmanager/alertmanager.yml +++ b/monitoring/alertmanager/alertmanager.yml @@ -15,6 +15,13 @@ route: group_interval: 5m repeat_interval: 4h routes: + # #1068: every critical alert opens/updates a tracked incident in the + # api-server (timeline, acknowledgment, postmortems). `continue` keeps the + # direct paging routes below as a backstop. + - matchers: ['severity="critical"'] + receiver: incident-tracker + group_wait: 0s + continue: true - matchers: ['severity="critical"'] receiver: pagerduty-oncall group_wait: 10s @@ -73,3 +80,21 @@ receivers: pagerduty_configs: - routing_key_file: /etc/alertmanager/secrets/pagerduty-key severity: critical + details: + firing: '{{ .Alerts.Firing | len }}' + runbook: '{{ .CommonAnnotations.runbook }}' + # #1068: Opsgenie alternative. Swap `receiver: pagerduty-oncall` above for + # `opsgenie-oncall` when INCIDENT_PROVIDER=opsgenie. + - name: opsgenie-oncall + opsgenie_configs: + - api_key_file: /etc/alertmanager/secrets/opsgenie-key + api_url: https://api.opsgenie.com/ + priority: '{{ if eq .CommonLabels.severity "critical" }}P1{{ else }}P3{{ end }}' + message: '{{ .CommonLabels.alertname }}: {{ .CommonAnnotations.summary }}' + tags: atomicip,{{ .CommonLabels.service }} + send_resolved: true + - name: incident-tracker + webhook_configs: + - url: http://atomicip-api.atomicip-production.svc/v1/admin/incidents/alertmanager + send_resolved: true + max_alerts: 50 diff --git a/monitoring/prometheus/operations-rules.yml b/monitoring/prometheus/operations-rules.yml new file mode 100644 index 0000000..baa09ed --- /dev/null +++ b/monitoring/prometheus/operations-rules.yml @@ -0,0 +1,91 @@ +# #1066 backup verification, #1067 cost optimization, #1069 GitOps sync alerts. +# Backup / cost metrics are pushed by scripts/ops/*.sh via the Pushgateway. +groups: + - name: backup-verification + rules: + - alert: BackupVerificationFailed + expr: atomicip_backup_verify_success == 0 + labels: + severity: critical + team: platform + service: backups + annotations: + summary: Latest backup restore test failed + runbook: docs/backup-strategy.md#5-when-verification-fails + - alert: BackupVerificationStale + expr: time() - atomicip_backup_verify_last_success_timestamp_seconds > 2 * 86400 + labels: + severity: critical + team: platform + service: backups + annotations: + summary: No successful backup restore test in over 48h + runbook: docs/backup-strategy.md#5-when-verification-fails + - alert: BackupTooOld + expr: atomicip_backup_age_seconds{kind="postgres_base"} > 26 * 3600 + or atomicip_backup_age_seconds{kind="postgres_wal"} > 900 + or atomicip_backup_age_seconds{kind="audit_log"} > 1800 + labels: + severity: warning + team: platform + service: backups + annotations: + summary: '{{ $labels.kind }} backup older than its RPO' + runbook: docs/backup-strategy.md#5-when-verification-fails + - alert: BackupRestoreSlowerThanRTO + expr: atomicip_backup_restore_seconds > 7200 + labels: + severity: warning + team: platform + service: backups + annotations: + summary: 'Database restore took {{ $value | humanizeDuration }} (RTO 2h)' + runbook: docs/backup-strategy.md#4-recovery-time + + - name: cost-optimization + rules: + - record: atomicip:cost_monthly_usd_saved:sum + expr: sum(atomicip_cost_monthly_usd_saved) + - record: atomicip:api_replicas:avg1d + expr: avg_over_time(kube_horizontalpodautoscaler_status_current_replicas{horizontalpodautoscaler="atomicip-api"}[1d]) + - alert: CostOptimizationJobStale + expr: time() - max(atomicip_cost_last_run_timestamp_seconds) > 3 * 86400 + labels: + severity: info + team: platform + annotations: + summary: Cost optimization job has not run for 3 days + runbook: docs/cost-optimization.md + - alert: ApiServerAtMaxReplicas + expr: | + kube_horizontalpodautoscaler_status_current_replicas{horizontalpodautoscaler="atomicip-api"} + >= kube_horizontalpodautoscaler_spec_max_replicas{horizontalpodautoscaler="atomicip-api"} + for: 30m + labels: + severity: warning + team: platform + service: api-server + annotations: + summary: api-server HPA pinned at maxReplicas for 30m; raise the cap or investigate load + runbook: docs/cost-optimization.md#1-auto-scaling + + - name: gitops + rules: + - alert: ArgoAppOutOfSync + expr: argocd_app_info{project="atomicip", sync_status!="Synced"} == 1 + for: 30m + labels: + severity: warning + team: platform + annotations: + summary: 'Argo CD application {{ $labels.name }} has been {{ $labels.sync_status }} for 30m' + runbook: docs/gitops.md#troubleshooting + - alert: ArgoAppDegraded + expr: argocd_app_info{project="atomicip", health_status="Degraded"} == 1 + for: 10m + labels: + severity: critical + team: platform + annotations: + summary: 'Argo CD application {{ $labels.name }} is Degraded' + runbook: docs/gitops.md#rollback diff --git a/scripts/gitops-rollback.sh b/scripts/gitops-rollback.sh new file mode 100755 index 0000000..ee56249 --- /dev/null +++ b/scripts/gitops-rollback.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# #1069: Roll back an environment by reverting its most recent promotion commit. +# +# scripts/gitops-rollback.sh <staging|production> [commit-sha] +# +# Creates a revert commit on a new branch and opens a PR (production) or pushes +# directly to main (staging). Argo CD then syncs the previous image tag. No +# kubectl access is required; the cluster always follows Git. +set -euo pipefail + +ENVIRONMENT="${1:?usage: $0 <staging|production> [commit-sha]}" +TARGET_SHA="${2:-}" +OVERLAY="deploy/k8s/overlays/${ENVIRONMENT}/kustomization.yaml" + +case "$ENVIRONMENT" in + staging|production) ;; + *) echo "unknown environment: $ENVIRONMENT" >&2; exit 2 ;; +esac + +git fetch origin main +if [ -z "$TARGET_SHA" ]; then + TARGET_SHA=$(git log origin/main -1 --format=%H -- "$OVERLAY") +fi +[ -n "$TARGET_SHA" ] || { echo "no commits touch $OVERLAY" >&2; exit 1; } + +echo "Reverting $(git log -1 --format='%h %s' "$TARGET_SHA")" + +BRANCH="gitops/rollback-${ENVIRONMENT}-${TARGET_SHA:0:12}" +git switch -c "$BRANCH" origin/main + +PARENTS=$(git rev-list --parents -n1 "$TARGET_SHA" | wc -w) +if [ "$PARENTS" -gt 2 ]; then + # Promotion PRs merged with a merge commit: revert relative to mainline. + git revert --no-edit -m 1 "$TARGET_SHA" +else + git revert --no-edit "$TARGET_SHA" +fi + +if [ "$ENVIRONMENT" = "production" ]; then + git push -u origin "$BRANCH" + if command -v gh >/dev/null 2>&1; then + gh pr create --base main --head "$BRANCH" \ + --title "rollback(production): revert ${TARGET_SHA:0:12}" \ + --label deployment,production,rollback \ + --body "Rolls production back by reverting ${TARGET_SHA}. Argo CD will sync the previous image once merged." + else + echo "Pushed $BRANCH; open a PR against main to complete the rollback." + fi +else + git push origin "HEAD:main" + echo "Staging rollback pushed; Argo CD will sync within its poll interval." +fi diff --git a/scripts/ops/backup-verify.sh b/scripts/ops/backup-verify.sh new file mode 100755 index 0000000..fc84871 --- /dev/null +++ b/scripts/ops/backup-verify.sh @@ -0,0 +1,210 @@ +#!/usr/bin/env bash +# #1066: Automated backup verification. +# +# Proves that the latest backups can actually be restored, not merely that they +# exist. Run on a schedule (deploy/k8s/jobs/backup-verification-cronjob.yaml and +# .github/workflows/backup-verification.yml). Steps: +# +# 1. Freshness latest Postgres base backup / WAL / audit archive are within RPO +# 2. Integrity SHA256SUMS + pg_verifybackup against the backup manifest +# 3. Restore restore base backup + replay WAL into a throwaway Postgres +# 4. Validation schema present, row counts sane, amcheck index verification +# 5. RTO time the restore and compare with the DR target +# 6. Report push metrics; on any failure fire an alert to Alertmanager +# +# Environment: +# BACKUP_BUCKET s3://bucket/prefix holding postgres/base, postgres/wal, audit/ +# PG_IMAGE Postgres image matching production (default postgres:16) +# RTO_TARGET_SECONDS DB restore target (default 7200, docs/disaster-recovery.md §2) +# BASE_BACKUP_MAX_AGE max age of newest base backup in seconds (default 93600 = 26h) +# WAL_MAX_AGE max age of newest WAL segment in seconds (default 300 = RPO 5 min) +# AUDIT_MAX_AGE max age of newest audit archive in seconds (default 1800) +# EXPECTED_TABLES space-separated tables that must exist (default below) +# PUSHGATEWAY_URL optional Prometheus Pushgateway +# ALERTMANAGER_URL optional Alertmanager (direct alert on failure) +# KEEP_WORKDIR=1 keep the restored data directory for debugging +set -euo pipefail + +: "${BACKUP_BUCKET:?BACKUP_BUCKET is required}" +PG_IMAGE="${PG_IMAGE:-postgres:16}" +RTO_TARGET_SECONDS="${RTO_TARGET_SECONDS:-7200}" +BASE_BACKUP_MAX_AGE="${BASE_BACKUP_MAX_AGE:-93600}" +WAL_MAX_AGE="${WAL_MAX_AGE:-300}" +AUDIT_MAX_AGE="${AUDIT_MAX_AGE:-1800}" +EXPECTED_TABLES="${EXPECTED_TABLES:-sessions webhooks webhook_deliveries ip_index audit_events}" +BUCKET="${BACKUP_BUCKET%/}" + +WORKDIR=$(mktemp -d -t atomicip-backup-verify-XXXXXX) +CONTAINER="atomicip-restore-$$" +STARTED=$(date +%s) +FAILURES=() +declare -A METRICS + +log() { printf '[backup-verify] %s\n' "$*" >&2; } +fail() { log "FAIL: $*"; FAILURES+=("$*"); } + +cleanup() { + docker rm -f "$CONTAINER" >/dev/null 2>&1 || true + [ "${KEEP_WORKDIR:-0}" = "1" ] && log "workdir kept at $WORKDIR" || rm -rf "$WORKDIR" +} +trap cleanup EXIT + +newest_object() { # prefix -> "epoch key" + aws s3 ls "$1" --recursive | sort -k1,2 | tail -n1 \ + | awk '{ cmd = "date -u -d \"" $1 " " $2 "\" +%s"; cmd | getline t; close(cmd); print t, $4 }' +} + +check_age() { # label prefix max_age + local newest ts key age + newest=$(newest_object "$2") + if [ -z "$newest" ]; then fail "$1: no backups found under $2"; return; fi + read -r ts key <<< "$newest" + age=$(( $(date +%s) - ts )) + METRICS["atomicip_backup_age_seconds{kind=\"$1\"}"]=$age + if [ "$age" -gt "$3" ]; then + fail "$1: newest backup $key is ${age}s old (limit $3s)" + else + log "$1: newest backup $key is ${age}s old" + fi +} + +# 1. Freshness -------------------------------------------------------------- +check_age postgres_base "$BUCKET/postgres/base/" "$BASE_BACKUP_MAX_AGE" +check_age postgres_wal "$BUCKET/postgres/wal/" "$WAL_MAX_AGE" +check_age audit_log "$BUCKET/audit/" "$AUDIT_MAX_AGE" + +# 2. Integrity -------------------------------------------------------------- +LATEST_BASE=$(aws s3 ls "$BUCKET/postgres/base/" | awk '/PRE/ {print $2}' | sort | tail -n1 | tr -d /) +if [ -z "$LATEST_BASE" ]; then + fail "no base backup directory found" +else + log "verifying base backup $LATEST_BASE" + DL_START=$(date +%s) + aws s3 cp --recursive --only-show-errors "$BUCKET/postgres/base/$LATEST_BASE/" "$WORKDIR/backup/" + METRICS[atomicip_backup_download_seconds]=$(( $(date +%s) - DL_START )) + + if [ -f "$WORKDIR/backup/SHA256SUMS" ]; then + (cd "$WORKDIR/backup" && sha256sum --quiet -c SHA256SUMS) \ + || fail "checksum mismatch in $LATEST_BASE" + else + fail "$LATEST_BASE has no SHA256SUMS" + fi + + mkdir -p "$WORKDIR/pgdata" + tar -xzf "$WORKDIR/backup/base.tar.gz" -C "$WORKDIR/pgdata" + [ -f "$WORKDIR/backup/pg_wal.tar.gz" ] && mkdir -p "$WORKDIR/pgdata/pg_wal" \ + && tar -xzf "$WORKDIR/backup/pg_wal.tar.gz" -C "$WORKDIR/pgdata/pg_wal" + cp "$WORKDIR/backup/backup_manifest" "$WORKDIR/pgdata/" 2>/dev/null \ + || fail "$LATEST_BASE has no backup_manifest" + + docker run --rm -v "$WORKDIR/pgdata:/pgdata" "$PG_IMAGE" \ + pg_verifybackup --no-parse-wal /pgdata >/dev/null \ + || fail "pg_verifybackup failed for $LATEST_BASE" +fi + +# 3. Restore + WAL replay --------------------------------------------------- +if [ -d "$WORKDIR/pgdata" ] && [ ${#FAILURES[@]} -eq 0 ]; then + RESTORE_START=$(date +%s) + mkdir -p "$WORKDIR/wal" + aws s3 sync --only-show-errors "$BUCKET/postgres/wal/" "$WORKDIR/wal/" + touch "$WORKDIR/pgdata/recovery.signal" + cat >> "$WORKDIR/pgdata/postgresql.auto.conf" <<CONF +restore_command = 'cp /wal/%f %p' +recovery_target_timeline = 'latest' +recovery_target_action = 'promote' +archive_mode = off +CONF + sudo_chown() { docker run --rm -v "$WORKDIR/pgdata:/pgdata" "$PG_IMAGE" chown -R postgres:postgres /pgdata; } + sudo_chown + + docker run -d --name "$CONTAINER" \ + -v "$WORKDIR/pgdata:/var/lib/postgresql/data" -v "$WORKDIR/wal:/wal:ro" \ + -e POSTGRES_HOST_AUTH_METHOD=trust "$PG_IMAGE" >/dev/null + + deadline=$(( $(date +%s) + RTO_TARGET_SECONDS )) + until docker exec "$CONTAINER" psql -U postgres -Atc "SELECT NOT pg_is_in_recovery()" 2>/dev/null | grep -q t; do + if [ "$(date +%s)" -gt "$deadline" ]; then fail "restore did not finish within RTO ${RTO_TARGET_SECONDS}s"; break; fi + if ! docker ps -q -f name="$CONTAINER" | grep -q .; then + docker logs "$CONTAINER" 2>&1 | tail -n 30 >&2 || true + fail "restored Postgres exited during recovery"; break + fi + sleep 5 + done + RESTORE_SECONDS=$(( $(date +%s) - RESTORE_START )) + METRICS[atomicip_backup_restore_seconds]=$RESTORE_SECONDS + log "restore + WAL replay took ${RESTORE_SECONDS}s (target ${RTO_TARGET_SECONDS}s)" + + # 4. Validation ----------------------------------------------------------- + if [ ${#FAILURES[@]} -eq 0 ]; then + q() { docker exec "$CONTAINER" psql -U postgres -d "${PGDATABASE:-atomicip}" -v ON_ERROR_STOP=1 -Atc "$1"; } + for t in $EXPECTED_TABLES; do + if [ "$(q "SELECT to_regclass('public.$t') IS NOT NULL")" != "t" ]; then + fail "table $t missing from restored database" + else + rows=$(q "SELECT count(*) FROM $t") + METRICS["atomicip_backup_restored_rows{table=\"$t\"}"]=$rows + fi + done + # Last committed transaction in the restore approximates achieved RPO. + last_xact=$(q "SELECT COALESCE(extract(epoch FROM pg_last_committed_xact_timestamp())::bigint, 0)" 2>/dev/null || echo 0) + [ "$last_xact" -gt 0 ] && METRICS[atomicip_backup_rpo_seconds]=$(( STARTED - last_xact )) + + # Structural integrity of every btree index (catches silent corruption). + q "CREATE EXTENSION IF NOT EXISTS amcheck" >/dev/null + bad=$(q "SELECT count(*) FROM ( + SELECT bt_index_check(c.oid) FROM pg_index i + JOIN pg_class c ON c.oid = i.indexrelid + JOIN pg_am a ON a.oid = c.relam + JOIN pg_namespace n ON n.oid = c.relnamespace + WHERE a.amname = 'btree' AND n.nspname = 'public' AND c.relpersistence <> 't') s" 2>&1) \ + || fail "amcheck reported index corruption: $bad" + fi + # 5. RTO ------------------------------------------------------------------ + [ "$RESTORE_SECONDS" -le "$RTO_TARGET_SECONDS" ] || fail "restore time ${RESTORE_SECONDS}s exceeds RTO ${RTO_TARGET_SECONDS}s" +fi + +# Audit log archive: newest object must decompress cleanly. +AUDIT_KEY=$(newest_object "$BUCKET/audit/" | awk '{print $2}') +if [ -n "$AUDIT_KEY" ]; then + aws s3 cp --only-show-errors "s3://$(echo "${BUCKET#s3://}" | cut -d/ -f1)/$AUDIT_KEY" "$WORKDIR/audit-sample" + case "$AUDIT_KEY" in + *.zst) zstd -t -q "$WORKDIR/audit-sample" || fail "audit archive $AUDIT_KEY is corrupt" ;; + *.gz) gzip -t "$WORKDIR/audit-sample" || fail "audit archive $AUDIT_KEY is corrupt" ;; + esac +fi + +# 6. Report ----------------------------------------------------------------- +TOTAL=$(( $(date +%s) - STARTED )) +SUCCESS=$([ ${#FAILURES[@]} -eq 0 ] && echo 1 || echo 0) +METRICS[atomicip_backup_verify_success]=$SUCCESS +METRICS[atomicip_backup_verify_duration_seconds]=$TOTAL +METRICS[atomicip_backup_verify_last_run_timestamp_seconds]=$(date +%s) +[ "$SUCCESS" = 1 ] && METRICS[atomicip_backup_verify_last_success_timestamp_seconds]=$(date +%s) + +if [ -n "${PUSHGATEWAY_URL:-}" ]; then + for k in "${!METRICS[@]}"; do printf '%s %s\n' "$k" "${METRICS[$k]}"; done \ + | curl -fsS --data-binary @- "${PUSHGATEWAY_URL}/metrics/job/backup_verification" >/dev/null \ + || log "pushgateway unreachable" +fi + +jq -n --arg base "${LATEST_BASE:-}" --argjson ok "$SUCCESS" --argjson total "$TOTAL" \ + --argjson restore "${RESTORE_SECONDS:-null}" --argjson rto "$RTO_TARGET_SECONDS" \ + --args '{base_backup:$base, success:($ok == 1), duration_seconds:$total, + restore_seconds:$restore, rto_target_seconds:$rto, failures:$ARGS.positional}' \ + "${FAILURES[@]}" > "${REPORT_OUT:-backup-verification.json}" + +if [ "$SUCCESS" = 1 ]; then + log "PASS: backup $LATEST_BASE restored in ${RESTORE_SECONDS:-?}s" + exit 0 +fi + +if [ -n "${ALERTMANAGER_URL:-}" ]; then + jq -cn --arg summary "Backup verification failed: ${FAILURES[0]}" \ + --arg desc "$(printf '%s; ' "${FAILURES[@]}")" \ + '[{labels:{alertname:"BackupVerificationFailed", severity:"critical", service:"backups", team:"platform"}, + annotations:{summary:$summary, description:$desc, runbook:"docs/backup-strategy.md#5-when-verification-fails"}}]' \ + | curl -fsS -H 'Content-Type: application/json' --data-binary @- \ + "${ALERTMANAGER_URL%/}/api/v2/alerts" >/dev/null || log "alertmanager unreachable" +fi +log "verification failed with ${#FAILURES[@]} problem(s)" +exit 1 diff --git a/scripts/ops/cost-optimize.sh b/scripts/ops/cost-optimize.sh new file mode 100755 index 0000000..3e3df37 --- /dev/null +++ b/scripts/ops/cost-optimize.sh @@ -0,0 +1,230 @@ +#!/usr/bin/env bash +# #1067: Cost optimization automation. +# +# scripts/ops/cost-optimize.sh cleanup # delete orphaned / expired off-chain data +# scripts/ops/cost-optimize.sh compress # compress + tier old audit logs and exports +# scripts/ops/cost-optimize.sh report # summarise savings for the last N days +# scripts/ops/cost-optimize.sh all # cleanup, compress, report +# +# Every action appends a JSON line to $COST_LEDGER describing what it saved, so +# savings are tracked over time and can be reported on. Metrics are also pushed +# to a Prometheus Pushgateway when PUSHGATEWAY_URL is set. +# +# Only off-chain caches/indexes are touched: the Stellar ledger is the source of +# truth (docs/disaster-recovery.md §1) and nothing here can affect it. +# +# Environment: +# DATABASE_URL Postgres connection string (required for cleanup) +# ARCHIVE_BUCKET s3://bucket/prefix for audit logs and exports +# AUDIT_LOG_DIR local audit log directory (default /var/log/atomicip/audit) +# RETENTION_DAYS orphan / expiry grace period (default 30) +# COMPRESS_AFTER_DAYS compress data older than this (default 7) +# COLD_TIER_AFTER_DAYS move to infrequent-access storage after this (default 90) +# STORAGE_COST_PER_GB $/GB-month hot storage (default 0.023) +# COLD_COST_PER_GB $/GB-month cold storage (default 0.0125) +# COST_LEDGER savings ledger (default /var/lib/atomicip/cost-ledger.jsonl) +# DRY_RUN=1 report what would be done without changing anything +set -euo pipefail + +RETENTION_DAYS="${RETENTION_DAYS:-30}" +COMPRESS_AFTER_DAYS="${COMPRESS_AFTER_DAYS:-7}" +COLD_TIER_AFTER_DAYS="${COLD_TIER_AFTER_DAYS:-90}" +STORAGE_COST_PER_GB="${STORAGE_COST_PER_GB:-0.023}" +COLD_COST_PER_GB="${COLD_COST_PER_GB:-0.0125}" +AUDIT_LOG_DIR="${AUDIT_LOG_DIR:-/var/log/atomicip/audit}" +COST_LEDGER="${COST_LEDGER:-/var/lib/atomicip/cost-ledger.jsonl}" +DRY_RUN="${DRY_RUN:-0}" + +log() { printf '[cost-optimize] %s\n' "$*" >&2; } + +run() { + if [ "$DRY_RUN" = "1" ]; then log "DRY_RUN: $*"; else "$@"; fi +} + +# record <action> <target> <items> <bytes_saved> <monthly_usd_saved> +record() { + mkdir -p "$(dirname "$COST_LEDGER")" + jq -cn --arg ts "$(date -u +%FT%TZ)" --arg action "$1" --arg target "$2" \ + --argjson items "$3" --argjson bytes "$4" --argjson usd "$5" \ + --argjson dry "$([ "$DRY_RUN" = "1" ] && echo true || echo false)" \ + '{ts:$ts, action:$action, target:$target, items:$items, bytes_saved:$bytes, monthly_usd_saved:$usd, dry_run:$dry}' \ + >> "$COST_LEDGER" + push_metric "$1" "$2" "$4" "$5" +} + +push_metric() { + [ -n "${PUSHGATEWAY_URL:-}" ] || return 0 + cat <<METRICS | curl -fsS --data-binary @- \ + "${PUSHGATEWAY_URL}/metrics/job/cost_optimization/action/$1/target/$2" >/dev/null || log "pushgateway unreachable" +# TYPE atomicip_cost_bytes_reclaimed gauge +atomicip_cost_bytes_reclaimed $3 +# TYPE atomicip_cost_monthly_usd_saved gauge +atomicip_cost_monthly_usd_saved $4 +# TYPE atomicip_cost_last_run_timestamp_seconds gauge +atomicip_cost_last_run_timestamp_seconds $(date +%s) +METRICS +} + +usd_for_bytes() { # bytes rate_per_gb + awk -v b="$1" -v r="$2" 'BEGIN { printf "%.4f", b / 1073741824 * r }' +} + +psql_q() { psql "$DATABASE_URL" -v ON_ERROR_STOP=1 -At -c "$1"; } + +# ---------------------------------------------------------------- cleanup --- + +# Each entry: target|size-estimate relation|WHERE clause selecting orphaned rows. +ORPHAN_RULES=( + "expired_sessions|sessions|expires_at < now() - interval '${RETENTION_DAYS} days'" + "orphaned_webhook_deliveries|webhook_deliveries|NOT EXISTS (SELECT 1 FROM webhooks w WHERE w.id = webhook_deliveries.webhook_id)" + "stale_webhook_deliveries|webhook_deliveries|delivered_at < now() - interval '${RETENTION_DAYS} days'" + "expired_recovery_tokens|recovery_tokens|expires_at < now() - interval '1 day'" + "orphaned_watchlist_entries|watchlist|created_at < now() - interval '${RETENTION_DAYS} days' AND NOT EXISTS (SELECT 1 FROM ip_index i WHERE i.ip_id = watchlist.ip_id)" +) + +cleanup_db() { + : "${DATABASE_URL:?DATABASE_URL is required for cleanup}" + local rule target table where exists rows avg_row bytes + for rule in "${ORPHAN_RULES[@]}"; do + IFS='|' read -r target table where <<< "$rule" + exists=$(psql_q "SELECT to_regclass('public.${table}') IS NOT NULL") + if [ "$exists" != "t" ]; then log "skip $target: table $table not present"; continue; fi + + rows=$(psql_q "SELECT count(*) FROM ${table} WHERE ${where}") + [ "$rows" -gt 0 ] || { log "$target: nothing to clean"; continue; } + avg_row=$(psql_q "SELECT COALESCE(pg_total_relation_size('${table}') / NULLIF((SELECT count(*) FROM ${table}), 0), 0)") + bytes=$(( rows * avg_row )) + + log "$target: deleting $rows rows (~$bytes bytes)" + if [ "$DRY_RUN" != "1" ]; then + # Batched deletes keep lock times and WAL bursts small. + while :; do + deleted=$(psql_q "WITH d AS (DELETE FROM ${table} WHERE ctid IN (SELECT ctid FROM ${table} WHERE ${where} LIMIT 5000) RETURNING 1) SELECT count(*) FROM d") + [ "$deleted" -gt 0 ] || break + done + psql_q "VACUUM (ANALYZE) ${table}" >/dev/null + fi + record cleanup "$target" "$rows" "$bytes" "$(usd_for_bytes "$bytes" "$STORAGE_COST_PER_GB")" + done +} + +cleanup_objects() { + [ -n "${ARCHIVE_BUCKET:-}" ] || { log "ARCHIVE_BUCKET unset; skipping object cleanup"; return 0; } + # Commitment exports (GET /ip/export) are ephemeral downloads. + local cutoff listing count bytes + cutoff=$(date -u -d "-${RETENTION_DAYS} days" +%F) + listing=$(aws s3 ls "${ARCHIVE_BUCKET%/}/exports/" --recursive | awk -v c="$cutoff" '$1 < c') + count=$(printf '%s' "$listing" | grep -c . || true) + bytes=$(printf '%s\n' "$listing" | awk '{s += $3} END {print s + 0}') + [ "$count" -gt 0 ] || { log "exports: nothing to clean"; return 0; } + log "exports: removing $count objects ($bytes bytes)" + printf '%s\n' "$listing" | awk '{print $4}' | while read -r key; do + run aws s3 rm "s3://$(echo "${ARCHIVE_BUCKET#s3://}" | cut -d/ -f1)/${key}" --only-show-errors + done + record cleanup stale_exports "$count" "$bytes" "$(usd_for_bytes "$bytes" "$STORAGE_COST_PER_GB")" +} + +# --------------------------------------------------------------- compress --- + +compress_audit_logs() { + [ -d "$AUDIT_LOG_DIR" ] || { log "no audit log dir at $AUDIT_LOG_DIR"; return 0; } + local count=0 saved=0 f before after + # The active log file is never touched; only rotated files (*.log.N / dated). + while IFS= read -r -d '' f; do + before=$(stat -c %s "$f") + if [ "$DRY_RUN" = "1" ]; then + after=$(zstd -19 -c "$f" | wc -c) + else + zstd -19 -q --rm "$f" -o "$f.zst" + # Audit logs are HMAC-chained (AUDIT_HMAC_KEY): verify the round-trip + # before trusting the compressed copy. + zstd -t -q "$f.zst" + after=$(stat -c %s "$f.zst") + fi + count=$((count + 1)); saved=$((saved + before - after)) + done < <(find "$AUDIT_LOG_DIR" -type f -name '*.log.*' ! -name '*.zst' ! -name '*.gz' \ + -mtime +"$COMPRESS_AFTER_DAYS" -print0) + [ "$count" -gt 0 ] || { log "audit logs: nothing to compress"; return 0; } + log "audit logs: compressed $count files, saved $saved bytes" + record compress audit_logs "$count" "$saved" "$(usd_for_bytes "$saved" "$STORAGE_COST_PER_GB")" +} + +compress_db_partitions() { + [ -n "${DATABASE_URL:-}" ] || return 0 + # Postgres 14+: switch old partitions of append-only tables to lz4 TOAST + # compression and rewrite them. Partition naming: <table>_yYYYYmMM. + local parts p before after count=0 saved=0 + parts=$(psql_q " + SELECT c.relname FROM pg_inherits i + JOIN pg_class c ON c.oid = i.inhrelid + JOIN pg_class p ON p.oid = i.inhparent + WHERE p.relname IN ('audit_events', 'chain_events') + AND to_date(substring(c.relname from 'y([0-9]{4}m[0-9]{2})$'), 'YYYY\"m\"MM') + < date_trunc('month', now() - interval '${COMPRESS_AFTER_DAYS} days') + AND NOT EXISTS (SELECT 1 FROM pg_description d WHERE d.objoid = c.oid AND d.description = 'compressed')" 2>/dev/null || true) + for p in $parts; do + before=$(psql_q "SELECT pg_total_relation_size('$p')") + if [ "$DRY_RUN" != "1" ]; then + psql_q "ALTER TABLE $p ALTER COLUMN payload SET COMPRESSION lz4" >/dev/null + psql_q "VACUUM (FULL, ANALYZE) $p" >/dev/null + psql_q "COMMENT ON TABLE $p IS 'compressed'" >/dev/null + fi + after=$(psql_q "SELECT pg_total_relation_size('$p')") + count=$((count + 1)); saved=$((saved + before - after)) + done + [ "$count" -gt 0 ] || return 0 + record compress db_partitions "$count" "$saved" "$(usd_for_bytes "$saved" "$STORAGE_COST_PER_GB")" +} + +tier_archive() { + [ -n "${ARCHIVE_BUCKET:-}" ] || return 0 + # Ensure the bucket moves old audit archives to cheaper storage classes. + # Object lock (retention) is unaffected by storage-class transitions. + local bucket prefix bytes + bucket=$(echo "${ARCHIVE_BUCKET#s3://}" | cut -d/ -f1) + prefix=$(echo "${ARCHIVE_BUCKET#s3://}" | cut -s -d/ -f2-) + run aws s3api put-bucket-lifecycle-configuration --bucket "$bucket" --lifecycle-configuration "$(jq -cn \ + --arg p "${prefix:+$prefix/}audit/" --argjson d "$COLD_TIER_AFTER_DAYS" --argjson g "$((COLD_TIER_AFTER_DAYS * 4))" \ + '{Rules:[{ID:"atomicip-audit-tiering",Status:"Enabled",Filter:{Prefix:$p}, + Transitions:[{Days:$d,StorageClass:"STANDARD_IA"},{Days:$g,StorageClass:"GLACIER_IR"}]}, + {ID:"atomicip-exports-expiry",Status:"Enabled",Filter:{Prefix:"exports/"},Expiration:{Days:30}}]}')" + bytes=$(aws s3 ls "${ARCHIVE_BUCKET%/}/audit/" --recursive --summarize | awk '/Total Size/ {print $3}') + bytes="${bytes:-0}" + record tier audit_archive 0 0 "$(awk -v b="$bytes" -v h="$STORAGE_COST_PER_GB" -v c="$COLD_COST_PER_GB" \ + 'BEGIN { printf "%.4f", b / 1073741824 * (h - c) }')" +} + +# ----------------------------------------------------------------- report --- + +report() { + local days="${REPORT_DAYS:-30}" since out + since=$(date -u -d "-${days} days" +%FT%TZ) + [ -f "$COST_LEDGER" ] || { log "no ledger at $COST_LEDGER"; return 0; } + out="${REPORT_OUT:-cost-report-$(date -u +%F).md}" + jq -rs --arg since "$since" --arg days "$days" ' + map(select(.ts >= $since and (.dry_run | not))) as $e + | ($e | group_by(.action + "/" + .target) + | map({key: (.[0].action + "/" + .[0].target), + runs: length, + items: (map(.items) | add), + gb: ((map(.bytes_saved) | add) / 1073741824), + usd: (map(.monthly_usd_saved) | add)})) as $rows + | "# AtomicIP cost optimization report\n\n" + + "Window: last \($days) days (since \($since))\n\n" + + "| Action / target | Runs | Items | GB reclaimed | Est. $/month saved |\n" + + "|---|---:|---:|---:|---:|\n" + + ($rows | map("| \(.key) | \(.runs) | \(.items) | \(.gb * 100 | round / 100) | \(.usd * 100 | round / 100) |") | join("\n")) + + "\n\n**Total reclaimed:** \(($e | map(.bytes_saved) | add // 0) / 1073741824 * 100 | round / 100) GB \n" + + "**Estimated monthly savings:** $\(($e | map(.monthly_usd_saved) | add // 0) * 100 | round / 100)\n" + ' "$COST_LEDGER" > "$out" + log "report written to $out" + cat "$out" +} + +case "${1:-all}" in + cleanup) cleanup_db; cleanup_objects ;; + compress) compress_audit_logs; compress_db_partitions; tier_archive ;; + report) report ;; + all) cleanup_db; cleanup_objects; compress_audit_logs; compress_db_partitions; tier_archive; report ;; + *) echo "usage: $0 {cleanup|compress|report|all}" >&2; exit 2 ;; +esac