diff --git a/k8s/prometheus-rules.yaml b/k8s/prometheus-rules.yaml index 3e56266..ee287c5 100644 --- a/k8s/prometheus-rules.yaml +++ b/k8s/prometheus-rules.yaml @@ -642,3 +642,39 @@ spec: Cross-correlate against audit_log kind=auth_probe_failed + the worker structured slog ERROR `auth_probe_failed leg=... reason=...`. Source: worker/internal/jobs/auth_probe.go. + + # Hourly synthetic deploy prober. Every 60 minutes the worker drives + # a real end-to-end deploy against prod (POST /deploy/new with + # redeploy=true → Kaniko build + k8s rollout → status poll until + # healthy → public-host GET expecting 200). Any fail outcome over a + # 30-minute window pages — at the 60-minute cadence a single fail + # IS already a regression, so the 30-minute window guarantees a + # P0 inside one tick rather than waiting two cycles (90 minutes of + # broken deploys before paging would be too long given the platform + # promises agent-driven deploys as a paid wedge). Mirrors + # deploy-probe-fail.json in newrelic/alerts/. + - name: instant-worker-deploy-probe + rules: + - alert: DeployProbeFail + expr: | + sum(increase(instant_deploy_probe_outcome_total{result="fail"}[30m])) > 0 + for: 30m + labels: + severity: critical + service: worker + annotations: + summary: "Hourly deploy prober fail (deploy pipeline broken in prod)" + description: | + instant_deploy_probe_outcome_total{result="fail"} > 0 in 30m. + The hourly deploy prober drives a real end-to-end deploy + against prod every 60 minutes. A fail outcome means one of: + /deploy/new returned non-2xx (api crashed, GHCR auth broken, + tier-limit miscount), the status row stayed `building` past + the 90s budget OR flipped to `failed` (Kaniko crash, + image-pull-backoff — the exact stuck-build class that hid + the 2026-05-30 morning truehomie-api incident for ~30 min), + or the public *.deployment.instanode.dev URL returned non-200 + after `healthy` (Ingress / TLS / pod-readiness regression). + Cross-correlate against audit_log kind=deploy_probe_failed + + the worker structured slog ERROR `deploy_probe_failed leg=... + reason=...`. Source: worker/internal/jobs/deploy_probe.go. diff --git a/newrelic/alerts/deploy-probe-fail.json b/newrelic/alerts/deploy-probe-fail.json new file mode 100644 index 0000000..05f80da --- /dev/null +++ b/newrelic/alerts/deploy-probe-fail.json @@ -0,0 +1,31 @@ +{ + "name": "instant-worker — deploy_probe_failed [synthetic deploy pipeline broken in prod]", + "type": "NRQL", + "description": "P0 page on ANY occurrence of instant_deploy_probe_outcome_total{result=\"fail\"}. The hourly deploy prober drives a real end-to-end deploy against the prod /deploy/new pipeline every 60 minutes (POST /deploy/new with redeploy=true → Kaniko build + k8s rollout → status poll until healthy → public-host GET expecting 200). A fail outcome means one of: (a) /deploy/new is returning non-2xx (api crashed, GHCR auth broken, tier-limit miscount), (b) the deploy row stays `building` past the 90s budget OR flips to `failed` (Kaniko crash, image-pull-backoff, the exact stuck-build class that hid the 2026-05-30 morning truehomie-api incident for ~30 minutes), or (c) the public *.deployment.instanode.dev URL returns non-200 after `healthy` (Ingress / TLS / pod-readiness regression). Cross-correlate against the audit_log row written by worker deploy_probe_failed (kind=deploy_probe_failed, actor='system:deploy_probe') AND the structured slog ERROR line deploy_probe_failed leg=... reason=... — same content on both surfaces. Source: worker/internal/jobs/deploy_probe.go (DeployProbeWorker), metric registered in worker/internal/metrics/metrics.go (DeployProbeOutcomeTotal). Threshold ABOVE 0 with 30m window: a single fail tick over 30 minutes is an unambiguous regression (two ticks at 60-minute cadence would be a full 90 minutes of broken deploys before paging — too long).", + "enabled": true, + "nrql": { + "query": "SELECT sum(instant_deploy_probe_outcome_total) FROM Metric WHERE metricName = 'instant_deploy_probe_outcome_total' AND result = 'fail'" + }, + "terms": [ + { + "priority": "CRITICAL", + "operator": "ABOVE", + "threshold": 0, + "thresholdDuration": 1800, + "thresholdOccurrences": "AT_LEAST_ONCE" + } + ], + "signal": { + "aggregationWindow": 60, + "aggregationMethod": "EVENT_FLOW", + "aggregationDelay": 120, + "fillOption": "STATIC", + "fillValue": 0 + }, + "expiration": { + "expirationDuration": 7200, + "openViolationOnExpiration": false, + "closeViolationsOnExpiration": true + }, + "violationTimeLimitSeconds": 86400 +} diff --git a/newrelic/dashboards/instanode-reliability.json b/newrelic/dashboards/instanode-reliability.json index 3082c16..35f079e 100644 --- a/newrelic/dashboards/instanode-reliability.json +++ b/newrelic/dashboards/instanode-reliability.json @@ -303,6 +303,51 @@ ], "platformOptions": { "ignoreTimeRange": false } } + }, + { + "title": "Hourly deploy prober — outcomes per leg (6h)", + "layout": { "column": 1, "row": 28, "width": 6, "height": 3 }, + "visualization": { "id": "viz.line" }, + "rawConfiguration": { + "nrqlQueries": [ + { + "accountIds": [0], + "query": "SELECT sum(instant_deploy_probe_outcome_total) FROM Metric WHERE metricName = 'instant_deploy_probe_outcome_total' FACET leg, result TIMESERIES SINCE 6 hours ago" + } + ], + "platformOptions": { "ignoreTimeRange": false } + } + }, + { + "title": "Hourly deploy prober — fails (last 6h, must be 0)", + "layout": { "column": 7, "row": 28, "width": 3, "height": 3 }, + "visualization": { "id": "viz.billboard" }, + "rawConfiguration": { + "nrqlQueries": [ + { + "accountIds": [0], + "query": "SELECT sum(instant_deploy_probe_outcome_total) AS 'fails' FROM Metric WHERE metricName = 'instant_deploy_probe_outcome_total' AND result = 'fail' SINCE 6 hours ago" + } + ], + "platformOptions": { "ignoreTimeRange": false }, + "thresholds": [ + { "alertSeverity": "CRITICAL", "value": 1 } + ] + } + }, + { + "title": "Hourly deploy prober — P95 latency per leg (6h)", + "layout": { "column": 10, "row": 28, "width": 3, "height": 3 }, + "visualization": { "id": "viz.line" }, + "rawConfiguration": { + "nrqlQueries": [ + { + "accountIds": [0], + "query": "SELECT percentile(instant_deploy_probe_latency_seconds, 95) AS 'p95' FROM Metric WHERE metricName = 'instant_deploy_probe_latency_seconds' FACET leg TIMESERIES SINCE 6 hours ago" + } + ], + "platformOptions": { "ignoreTimeRange": false } + } } ] } diff --git a/observability/METRICS-CATALOG.md b/observability/METRICS-CATALOG.md index ebcce2b..a9ed023 100644 --- a/observability/METRICS-CATALOG.md +++ b/observability/METRICS-CATALOG.md @@ -41,6 +41,8 @@ fires. Operators need this so they don't panic when a fresh deploy looks | `instant_idempotency_replay_refunded_total` | api | `route` | lazy (CounterVec — first cache HIT on each route materialises the label series; a fresh deploy with no retries reports nothing until the first agent retries with the same `Idempotency-Key`) | `idempotency-replay-refund-spike.json` | `IdempotencyReplayRefundSpike` | "Idempotency replay refunds by route (1h) — FINDING API-1" | | `instant_auth_probe_outcome_total` | worker | `leg,result` | lazy (CounterVec — `pass`/`degraded` materialise on the first happy tick; `fail` only appears after a real regression. AUTH-004 synthetic prober: every 5 min the worker drives /auth/email/start + /auth/exchange CORS contract + /auth/me bearer against prod) | `auth-probe-fail.json` | `AuthProbeFail` | "AUTH-004 synthetic prober — outcomes per leg (1h)", "AUTH-004 synthetic prober — fails (last 1h, must be 0)" | | `instant_auth_probe_latency_seconds` | worker | `leg` | lazy (HistogramVec — observation only on a real HTTP response; DNS/TCP errors omit the observation so the histogram isn't polluted with 0s timeouts) | (covered by `auth-probe-fail.json`) | (covered by `AuthProbeFail`) | "AUTH-004 synthetic prober — P95 latency per leg (1h)" | +| `instant_deploy_probe_outcome_total` | worker | `leg,result` | lazy (CounterVec — `pass`/`degraded` materialise on the first happy tick; `fail` only appears after a real regression. Hourly deploy prober: every 60 min the worker drives /deploy/new + status-poll until healthy + public-host GET against prod. Closes the 2026-05-30 stuck-build gap that hid a broken deploy pipeline for ~30 min) | `deploy-probe-fail.json` | `DeployProbeFail` | "Hourly deploy prober — outcomes per leg (6h)", "Hourly deploy prober — fails (last 6h, must be 0)" | +| `instant_deploy_probe_latency_seconds` | worker | `leg` | lazy (HistogramVec — observation only on a real HTTP response or successful status flip; DNS/TCP errors omit the observation. Buckets span the per-leg budgets up to the 120s cold-cluster Kaniko ceiling) | (covered by `deploy-probe-fail.json`) | (covered by `DeployProbeFail`) | "Hourly deploy prober — P95 latency per leg (6h)" | ## Lazy-emit gotcha — what operators should expect